Update patches for 8.1.1

Signed-off-by: nyanmisaka <nst799610810@gmail.com>
This commit is contained in:
nyanmisaka
2026-05-28 10:53:47 +08:00
parent 959574d802
commit 6f10ceb986
107 changed files with 6058 additions and 6563 deletions
+3 -2
View File
@@ -165,7 +165,7 @@ Index: FFmpeg/compat/cuda/cuda_runtime.h
{ {
float4 ret; float4 ret;
asm("tex.2d.v4.f32.f32 {%0, %1, %2, %3}, [%4, {%5, %6}];" : asm("tex.2d.v4.f32.f32 {%0, %1, %2, %3}, [%4, {%5, %6}];" :
@@ -156,37 +147,42 @@ inline __device__ float4 tex2D<float4>(c @@ -156,38 +147,43 @@ inline __device__ float4 tex2D<float4>(c
} }
template<> template<>
@@ -228,4 +228,5 @@ Index: FFmpeg/compat/cuda/cuda_runtime.h
+static __inline__ __device__ float __log10f(float a) { return __nvvm_lg2_approx_f(a) * 0.30103f; } +static __inline__ __device__ float __log10f(float a) { return __nvvm_lg2_approx_f(a) * 0.30103f; }
+static __inline__ __device__ float __powf(float a, float b) { return __nvvm_ex2_approx_f(__nvvm_lg2_approx_f(a) * b); } +static __inline__ __device__ float __powf(float a, float b) { return __nvvm_ex2_approx_f(__nvvm_lg2_approx_f(a) * b); }
#endif /* COMPAT_CUDA_CUDA_RUNTIME_H */ // Misc helper functions
extern "C" __device__ int printf(const char*, ...);
@@ -327,7 +327,7 @@ Index: FFmpeg/libavfilter/vf_scale_cuda.c
#include "filters.h" #include "filters.h"
#include "scale_eval.h" #include "scale_eval.h"
#include "video.h" #include "video.h"
@@ -108,6 +110,9 @@ typedef struct CUDAScaleContext { @@ -121,6 +123,9 @@ typedef struct CUDAScaleContext {
int interp_as_integer; int interp_as_integer;
float param; float param;
@@ -337,7 +337,7 @@ Index: FFmpeg/libavfilter/vf_scale_cuda.c
} CUDAScaleContext; } CUDAScaleContext;
static av_cold int cudascale_init(AVFilterContext *ctx) static av_cold int cudascale_init(AVFilterContext *ctx)
@@ -129,13 +134,23 @@ static av_cold void cudascale_uninit(AVF @@ -142,13 +147,23 @@ static av_cold void cudascale_uninit(AVF
{ {
CUDAScaleContext *s = ctx->priv; CUDAScaleContext *s = ctx->priv;
@@ -364,7 +364,7 @@ Index: FFmpeg/libavfilter/vf_scale_cuda.c
CHECK_CU(cu->cuCtxPopCurrent(&dummy)); CHECK_CU(cu->cuCtxPopCurrent(&dummy));
} }
@@ -275,6 +290,69 @@ static av_cold int init_processing_chain @@ -294,6 +309,67 @@ static av_cold int init_processing_chain
return 0; return 0;
} }
@@ -387,30 +387,28 @@ Index: FFmpeg/libavfilter/vf_scale_cuda.c
+ .srcPitch = ff_fruit_dither_size * sizeof(ff_fruit_dither_matrix[0]), + .srcPitch = ff_fruit_dither_size * sizeof(ff_fruit_dither_matrix[0]),
+ .dstPitch = ff_fruit_dither_size * sizeof(ff_fruit_dither_matrix[0]), + .dstPitch = ff_fruit_dither_size * sizeof(ff_fruit_dither_matrix[0]),
+ .WidthInBytes = ff_fruit_dither_size * sizeof(ff_fruit_dither_matrix[0]), + .WidthInBytes = ff_fruit_dither_size * sizeof(ff_fruit_dither_matrix[0]),
+ .Height = ff_fruit_dither_size, + .Height = ff_fruit_dither_size
+ }; + };
+ +
+#ifndef CU_TRSF_NORMALIZED_COORDINATES
+ #define CU_TRSF_NORMALIZED_COORDINATES 2
+#endif
+ CUDA_TEXTURE_DESC tex_desc = { + CUDA_TEXTURE_DESC tex_desc = {
+ .addressMode = { CU_TR_ADDRESS_MODE_WRAP, + .addressMode = { CU_TR_ADDRESS_MODE_WRAP,
+ CU_TR_ADDRESS_MODE_WRAP }, + CU_TR_ADDRESS_MODE_WRAP },
+ .filterMode = CU_TR_FILTER_MODE_POINT, + .filterMode = CU_TR_FILTER_MODE_POINT,
+ .flags = CU_TRSF_NORMALIZED_COORDINATES, + .flags = 2 /* CU_TRSF_NORMALIZED_COORDINATES */
+ }; + };
+ +
+ CUDA_RESOURCE_DESC res_desc = { + CUDA_RESOURCE_DESC res_desc = {
+ .resType = CU_RESOURCE_TYPE_PITCH2D, + .resType = CU_RESOURCE_TYPE_PITCH2D,
+ .res.pitch2D.format = CU_AD_FORMAT_UNSIGNED_INT16, + .res.pitch2D.format = CU_AD_FORMAT_UNSIGNED_INT16,
+ .res.pitch2D.numChannels = 1, + .res.pitch2D.numChannels = 1,
+ .res.pitch2D.width = ff_fruit_dither_size, + .res.pitch2D.width = ff_fruit_dither_size,
+ .res.pitch2D.height = ff_fruit_dither_size, + .res.pitch2D.height = ff_fruit_dither_size,
+ .res.pitch2D.pitchInBytes = ff_fruit_dither_size * sizeof(ff_fruit_dither_matrix[0]), + .res.pitch2D.pitchInBytes = ff_fruit_dither_size * sizeof(ff_fruit_dither_matrix[0]),
+ .res.pitch2D.devPtr = 0, + .res.pitch2D.devPtr = 0
+ }; + };
+ +
+ av_assert0(sizeof(ff_fruit_dither_matrix) == sizeof(ff_fruit_dither_matrix[0]) * ff_fruit_dither_size * ff_fruit_dither_size); + av_assert0(sizeof(ff_fruit_dither_matrix) ==
+ sizeof(ff_fruit_dither_matrix[0]) * ff_fruit_dither_size * ff_fruit_dither_size);
+ +
+ if ((ret = CHECK_CU(cu->cuCtxPushCurrent(cuda_ctx))) < 0) + if ((ret = CHECK_CU(cu->cuCtxPushCurrent(cuda_ctx))) < 0)
+ return ret; + return ret;
@@ -434,7 +432,7 @@ Index: FFmpeg/libavfilter/vf_scale_cuda.c
static av_cold int cudascale_load_functions(AVFilterContext *ctx) static av_cold int cudascale_load_functions(AVFilterContext *ctx)
{ {
CUDAScaleContext *s = ctx->priv; CUDAScaleContext *s = ctx->priv;
@@ -383,6 +461,11 @@ static av_cold int cudascale_config_prop @@ -409,6 +485,11 @@ static av_cold int cudascale_config_prop
s->hwctx = device_hwctx; s->hwctx = device_hwctx;
s->cu_stream = s->hwctx->stream; s->cu_stream = s->hwctx->stream;
@@ -443,27 +441,22 @@ Index: FFmpeg/libavfilter/vf_scale_cuda.c
+ goto fail; + goto fail;
+ } + }
+ +
if (inlink->sample_aspect_ratio.num) { if (s->reset_sar)
outlink->sample_aspect_ratio = av_mul_q((AVRational){outlink->h*inlink->w, outlink->sample_aspect_ratio = (AVRational){1, 1};
outlink->w*inlink->h}, else if (inlink->sample_aspect_ratio.num) {
@@ -418,11 +501,15 @@ static int call_resize_kernel(AVFilterCo @@ -457,7 +538,10 @@ static int call_resize_kernel(AVFilterCo
(CUdeviceptr)out_frame->data[2], (CUdeviceptr)out_frame->data[3] .src_width = src_width,
.src_height = src_height,
.param = s->param,
- .mpeg_range = mpeg_range
+ .mpeg_range = mpeg_range,
+ .dither_tex = s->dither_tex,
+ .dither_size = (float)ff_fruit_dither_size,
+ .dither_quantization = (float)((1 << s->out_desc->comp[0].depth) - 1)
}; };
+ float dither_size = (float)ff_fruit_dither_size; void *args[] = { &params };
+ float dither_quantization = (float)((1 << s->out_desc->comp[0].depth) - 1); @@ -487,6 +571,7 @@ static int scalecuda_resize(AVFilterCont
+
void *args_uchar[] = {
&src_tex[0], &src_tex[1], &src_tex[2], &src_tex[3],
&dst_devptr[0], &dst_devptr[1], &dst_devptr[2], &dst_devptr[3],
&dst_width, &dst_height, &dst_pitch,
- &src_width, &src_height, &s->param
+ &src_width, &src_height, &s->param,
+ &s->dither_tex, &dither_size, &dither_quantization
};
return CHECK_CU(cu->cuLaunchKernel(func,
@@ -446,6 +533,7 @@ static int scalecuda_resize(AVFilterCont
for (i = 0; i < s->in_planes; i++) { for (i = 0; i < s->in_planes; i++) {
CUDA_TEXTURE_DESC tex_desc = { CUDA_TEXTURE_DESC tex_desc = {
@@ -475,7 +468,7 @@ Index: FFmpeg/libavfilter/vf_scale_cuda.cu
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/vf_scale_cuda.cu --- FFmpeg.orig/libavfilter/vf_scale_cuda.cu
+++ FFmpeg/libavfilter/vf_scale_cuda.cu +++ FFmpeg/libavfilter/vf_scale_cuda.cu
@@ -29,6 +29,19 @@ using subsample_function_t = T (*)(cudaT @@ -30,6 +30,19 @@ using subsample_function_t = T (*)(cudaT
int src_width, int src_height, int src_width, int src_height,
int bit_depth, float param); int bit_depth, float param);
@@ -495,54 +488,42 @@ Index: FFmpeg/libavfilter/vf_scale_cuda.cu
// --- CONVERSION LOGIC --- // --- CONVERSION LOGIC ---
static const ushort mask_10bit = 0xFFC0; static const ushort mask_10bit = 0xFFC0;
@@ -64,7 +77,9 @@ static inline __device__ ushort conv_16t @@ -90,7 +103,9 @@ static inline __device__ ushort conv_16t
subsample_function_t<in_T_uv> subsample_func_uv> \
__device__ static inline void N(cudaTextureObject_t src_tex[4], T *dst[4], int xo, int yo, \ __device__ static inline void N(cudaTextureObject_t src_tex[4], T *dst[4], int xo, int yo, \
int dst_width, int dst_height, int dst_pitch, \ int dst_width, int dst_height, int dst_pitch, \
- int src_width, int src_height, float param) int src_left, int src_top, int src_width, int src_height, \
+ int src_width, int src_height, float param, \ - float param, int mpeg_range)
+ float param, int mpeg_range, \
+ cudaTextureObject_t dither_tex, \ + cudaTextureObject_t dither_tex, \
+ float dither_size, float dither_quantization) + float dither_size, float dither_quantization)
#define SUB_F(m, plane) \ #define SUB_F(m, plane) \
subsample_func_##m(src_tex[plane], xo, yo, \ subsample_func_##m(src_tex[plane], xo, yo, \
@@ -477,7 +492,10 @@ struct Convert_p010le_yuv420p @@ -248,7 +263,10 @@ struct Convert_planar10_planar8
DEF_F(Convert, out_T) DEF_F(Convert, out_T)
{ {
- DEFAULT_DST(0) = conv_10to8(SUB_F(y, 0)); - DEFAULT_DST(0) = conv_10to8pl(SUB_F(y, 0));
+ in_T res = SUB_F(y, 0); + in_T res = SUB_F(y, 0);
+ res = (in_T)get_dithered_y((float)res, read_dither(dither_tex, dither_size, xo, yo), + res = (in_T)get_dithered_y((float)res, read_dither(dither_tex, dither_size, xo, yo),
+ dither_size, dither_quantization, (float)mask_10bit); + dither_size, dither_quantization, (float)0x3FF);
+ DEFAULT_DST(0) = conv_10to8(res); + DEFAULT_DST(0) = conv_10to8pl(res);
} }
DEF_F(Convert_uv, out_T_uv) DEF_F(Convert_uv, out_T_uv)
@@ -498,7 +516,10 @@ struct Convert_p010le_nv12 @@ -308,7 +326,10 @@ struct Convert_planar10_semiplanar8
DEF_F(Convert, out_T) DEF_F(Convert, out_T)
{ {
- DEFAULT_DST(0) = conv_10to8(SUB_F(y, 0)); - DEFAULT_DST(0) = conv_10to8pl(SUB_F(y, 0));
+ in_T res = SUB_F(y, 0); + in_T res = SUB_F(y, 0);
+ res = (in_T)get_dithered_y((float)res, read_dither(dither_tex, dither_size, xo, yo), + res = (in_T)get_dithered_y((float)res, read_dither(dither_tex, dither_size, xo, yo),
+ dither_size, dither_quantization, (float)mask_10bit); + dither_size, dither_quantization, (float)0x3FF);
+ DEFAULT_DST(0) = conv_10to8(res); + DEFAULT_DST(0) = conv_10to8pl(res);
} }
DEF_F(Convert_uv, out_T_uv) DEF_F(Convert_uv, out_T_uv)
@@ -521,7 +542,10 @@ struct Convert_p010le_yuv444p @@ -376,7 +397,10 @@ struct Convert_planar16_planar8
DEF_F(Convert, out_T)
{
- DEFAULT_DST(0) = conv_10to8(SUB_F(y, 0));
+ in_T res = SUB_F(y, 0);
+ res = (in_T)get_dithered_y((float)res, read_dither(dither_tex, dither_size, xo, yo),
+ dither_size, dither_quantization, (float)mask_10bit);
+ DEFAULT_DST(0) = conv_10to8(res);
}
DEF_F(Convert_uv, out_T_uv)
@@ -607,7 +631,10 @@ struct Convert_p016le_yuv420p
DEF_F(Convert, out_T) DEF_F(Convert, out_T)
{ {
@@ -554,7 +535,19 @@ Index: FFmpeg/libavfilter/vf_scale_cuda.cu
} }
DEF_F(Convert_uv, out_T_uv) DEF_F(Convert_uv, out_T_uv)
@@ -628,7 +655,10 @@ struct Convert_p016le_nv12 @@ -396,7 +420,10 @@ struct Convert_planar16_planar10
DEF_F(Convert, out_T)
{
- DEFAULT_DST(0) = conv_16to10pl(SUB_F(y, 0));
+ in_T res = SUB_F(y, 0);
+ res = (in_T)get_dithered_y((float)res, read_dither(dither_tex, dither_size, xo, yo),
+ dither_size, dither_quantization, (float)mask_16bit);
+ DEFAULT_DST(0) = conv_16to10pl(res);
}
DEF_F(Convert_uv, out_T_uv)
@@ -436,7 +463,10 @@ struct Convert_planar16_semiplanar8
DEF_F(Convert, out_T) DEF_F(Convert, out_T)
{ {
@@ -566,19 +559,7 @@ Index: FFmpeg/libavfilter/vf_scale_cuda.cu
} }
DEF_F(Convert_uv, out_T_uv) DEF_F(Convert_uv, out_T_uv)
@@ -651,7 +681,10 @@ struct Convert_p016le_yuv444p @@ -458,7 +488,10 @@ struct Convert_planar16_semiplanar10
DEF_F(Convert, out_T)
{
- DEFAULT_DST(0) = conv_16to8(SUB_F(y, 0));
+ in_T res = SUB_F(y, 0);
+ res = (in_T)get_dithered_y((float)res, read_dither(dither_tex, dither_size, xo, yo),
+ dither_size, dither_quantization, (float)mask_16bit);
+ DEFAULT_DST(0) = conv_16to8(res);
}
DEF_F(Convert_uv, out_T_uv)
@@ -672,7 +705,10 @@ struct Convert_p016le_p010le
DEF_F(Convert, out_T) DEF_F(Convert, out_T)
{ {
@@ -590,7 +571,31 @@ Index: FFmpeg/libavfilter/vf_scale_cuda.cu
} }
DEF_F(Convert_uv, out_T_uv) DEF_F(Convert_uv, out_T_uv)
@@ -737,7 +773,10 @@ struct Convert_yuv444p16le_yuv420p @@ -634,7 +667,10 @@ struct Convert_semiplanar10_planar8
DEF_F(Convert, out_T)
{
- DEFAULT_DST(0) = conv_10to8(SUB_F(y, 0));
+ in_T res = SUB_F(y, 0);
+ res = (in_T)get_dithered_y((float)res, read_dither(dither_tex, dither_size, xo, yo),
+ dither_size, dither_quantization, (float)mask_10bit);
+ DEFAULT_DST(0) = conv_10to8(res);
}
DEF_F(Convert_uv, out_T_uv)
@@ -697,7 +733,10 @@ struct Convert_semiplanar10_semiplanar8
DEF_F(Convert, out_T)
{
- DEFAULT_DST(0) = conv_10to8(SUB_F(y, 0));
+ in_T res = SUB_F(y, 0);
+ res = (in_T)get_dithered_y((float)res, read_dither(dither_tex, dither_size, xo, yo),
+ dither_size, dither_quantization, (float)mask_10bit);
+ DEFAULT_DST(0) = conv_10to8(res);
}
DEF_F(Convert_uv, out_T_uv)
@@ -765,7 +804,10 @@ struct Convert_semiplanar16_planar8
DEF_F(Convert, out_T) DEF_F(Convert, out_T)
{ {
@@ -602,7 +607,19 @@ Index: FFmpeg/libavfilter/vf_scale_cuda.cu
} }
DEF_F(Convert_uv, out_T_uv) DEF_F(Convert_uv, out_T_uv)
@@ -757,7 +796,10 @@ struct Convert_yuv444p16le_nv12 @@ -786,7 +828,10 @@ struct Convert_semiplanar16_planar10
DEF_F(Convert, out_T)
{
- DEFAULT_DST(0) = conv_16to10pl(SUB_F(y, 0));
+ in_T res = SUB_F(y, 0);
+ res = (in_T)get_dithered_y((float)res, read_dither(dither_tex, dither_size, xo, yo),
+ dither_size, dither_quantization, (float)mask_16bit);
+ DEFAULT_DST(0) = conv_16to10pl(res);
}
DEF_F(Convert_uv, out_T_uv)
@@ -828,7 +873,10 @@ struct Convert_semiplanar16_semiplanar8
DEF_F(Convert, out_T) DEF_F(Convert, out_T)
{ {
@@ -614,19 +631,7 @@ Index: FFmpeg/libavfilter/vf_scale_cuda.cu
} }
DEF_F(Convert_uv, out_T_uv) DEF_F(Convert_uv, out_T_uv)
@@ -779,7 +821,10 @@ struct Convert_yuv444p16le_yuv444p @@ -851,7 +899,10 @@ struct Convert_semiplanar16_semiplanar10
DEF_F(Convert, out_T)
{
- DEFAULT_DST(0) = conv_16to8(SUB_F(y, 0));
+ in_T res = SUB_F(y, 0);
+ res = (in_T)get_dithered_y((float)res, read_dither(dither_tex, dither_size, xo, yo),
+ dither_size, dither_quantization, (float)mask_16bit);
+ DEFAULT_DST(0) = conv_16to8(res);
}
DEF_F(Convert_uv, out_T_uv)
@@ -799,7 +844,10 @@ struct Convert_yuv444p16le_p010le
DEF_F(Convert, out_T) DEF_F(Convert, out_T)
{ {
@@ -638,7 +643,7 @@ Index: FFmpeg/libavfilter/vf_scale_cuda.cu
} }
DEF_F(Convert_uv, out_T_uv) DEF_F(Convert_uv, out_T_uv)
@@ -1015,6 +1063,9 @@ typedef float4 (*coeffs_function_t)(floa @@ -1045,6 +1096,9 @@ typedef float4 (*coeffs_function_t)(floa
__device__ static inline float4 lanczos_coeffs(float x, float param) __device__ static inline float4 lanczos_coeffs(float x, float param)
{ {
@@ -648,7 +653,7 @@ Index: FFmpeg/libavfilter/vf_scale_cuda.cu
const float pi = 3.141592654f; const float pi = 3.141592654f;
float4 res = make_float4( float4 res = make_float4(
@@ -1129,12 +1180,15 @@ __device__ static inline T Subsample_Bic @@ -1162,12 +1216,15 @@ __device__ static inline T Subsample_Bic
#define PIX(x, y) tex2D<floatT>(tex, (x), (y)) #define PIX(x, y) tex2D<floatT>(tex, (x), (y))
return from_floatN<T, floatT>( return from_floatN<T, floatT>(
@@ -670,25 +675,28 @@ Index: FFmpeg/libavfilter/vf_scale_cuda.cu
); );
#undef PIX #undef PIX
@@ -1147,7 +1201,9 @@ __device__ static inline T Subsample_Bic @@ -1194,7 +1251,8 @@ __device__ static inline T Subsample_Bic
cudaTextureObject_t src_tex_2, cudaTextureObject_t src_tex_3, \ params.dst_width, params.dst_height, params.dst_pitch, \
T *dst_0, T *dst_1, T *dst_2, T *dst_3, \ params.src_left, params.src_top, \
int dst_width, int dst_height, int dst_pitch, \ params.src_width, params.src_height, \
- int src_width, int src_height, float param - params.param, params.mpeg_range);
+ int src_width, int src_height, float param, \ + params.param, params.mpeg_range, \
+ cudaTextureObject_t dither_tex, \ + params.dither_tex, params.dither_size, params.dither_quantization);
+ float dither_size, float dither_quantization
#define SUBSAMPLE(Convert, T) \
cudaTextureObject_t src_tex[4] = \
@@ -1159,7 +1215,9 @@ __device__ static inline T Subsample_Bic
Convert( \
src_tex, dst, xo, yo, \
dst_width, dst_height, dst_pitch, \
- src_width, src_height, param);
+ src_width, src_height, param, \
+ dither_tex, \
+ dither_size, dither_quantization);
extern "C" { extern "C" {
Index: FFmpeg/libavfilter/vf_scale_cuda.h
===================================================================
--- FFmpeg.orig/libavfilter/vf_scale_cuda.h
+++ FFmpeg/libavfilter/vf_scale_cuda.h
@@ -45,6 +45,10 @@ typedef struct {
int src_height;
float param;
int mpeg_range;
+
+ CUtexObject dither_tex;
+ float dither_size;
+ float dither_quantization;
} CUDAScaleKernelParams;
#endif
+160 -151
View File
@@ -2,7 +2,7 @@ Index: FFmpeg/configure
=================================================================== ===================================================================
--- FFmpeg.orig/configure --- FFmpeg.orig/configure
+++ FFmpeg/configure +++ FFmpeg/configure
@@ -3316,6 +3316,8 @@ scale_cuda_filter_deps="ffnvcodec" @@ -3522,6 +3522,8 @@ scale_cuda_filter_deps="ffnvcodec"
scale_cuda_filter_deps_any="cuda_nvcc cuda_llvm" scale_cuda_filter_deps_any="cuda_nvcc cuda_llvm"
thumbnail_cuda_filter_deps="ffnvcodec" thumbnail_cuda_filter_deps="ffnvcodec"
thumbnail_cuda_filter_deps_any="cuda_nvcc cuda_llvm" thumbnail_cuda_filter_deps_any="cuda_nvcc cuda_llvm"
@@ -11,9 +11,9 @@ Index: FFmpeg/configure
transpose_npp_filter_deps="ffnvcodec libnpp" transpose_npp_filter_deps="ffnvcodec libnpp"
overlay_cuda_filter_deps="ffnvcodec" overlay_cuda_filter_deps="ffnvcodec"
overlay_cuda_filter_deps_any="cuda_nvcc cuda_llvm" overlay_cuda_filter_deps_any="cuda_nvcc cuda_llvm"
@@ -4707,6 +4709,15 @@ if enabled cuda_nvcc; then @@ -4987,6 +4989,15 @@ if enabled cuda_nvcc; then
if $nvcc $nvccflags_default 2>&1 | grep -qi unsupported; then if $nvcc $nvccflags_default 2>&1 | grep -qi unsupported; then
nvccflags_default="-gencode arch=compute_60,code=sm_60 -O2" nvccflags_default="-gencode arch=compute_75,code=sm_75 -O2"
fi fi
+else +else
+ clang_version=$($nvcc -dumpversion 2>/dev/null || echo 0) + clang_version=$($nvcc -dumpversion 2>/dev/null || echo 0)
@@ -27,7 +27,7 @@ Index: FFmpeg/configure
fi fi
set_default arch cc cxx doxygen pkg_config ranlib strip sysinclude \ set_default arch cc cxx doxygen pkg_config ranlib strip sysinclude \
@@ -6774,7 +6785,7 @@ fi @@ -7117,7 +7128,7 @@ fi
if enabled cuda_nvcc; then if enabled cuda_nvcc; then
nvccflags="$nvccflags -ptx" nvccflags="$nvccflags -ptx"
else else
@@ -40,9 +40,9 @@ Index: FFmpeg/ffbuild/common.mak
=================================================================== ===================================================================
--- FFmpeg.orig/ffbuild/common.mak --- FFmpeg.orig/ffbuild/common.mak
+++ FFmpeg/ffbuild/common.mak +++ FFmpeg/ffbuild/common.mak
@@ -44,6 +44,7 @@ ASFLAGS := $(CPPFLAGS) $(ASFLAGS) @@ -53,6 +53,7 @@ ASFLAGS := $(CPPFLAGS) $(ASFLAGS)
# end up in CXXFLAGS. # end up in CXXFLAGS.
$(call PREPEND,CXXFLAGS, CPPFLAGS CFLAGS) $(call PREPEND,CXXFLAGS, CPPFLAGS)
X86ASMFLAGS += $(IFLAGS:%=%/) -I$(<D)/ -Pconfig.asm X86ASMFLAGS += $(IFLAGS:%=%/) -I$(<D)/ -Pconfig.asm
+NVCCFLAGS += $(IFLAGS) +NVCCFLAGS += $(IFLAGS)
@@ -52,7 +52,7 @@ Index: FFmpeg/libavfilter/Makefile
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/Makefile --- FFmpeg.orig/libavfilter/Makefile
+++ FFmpeg/libavfilter/Makefile +++ FFmpeg/libavfilter/Makefile
@@ -524,6 +524,8 @@ OBJS-$(CONFIG_TMEDIAN_FILTER) @@ -541,6 +541,8 @@ OBJS-$(CONFIG_TMEDIAN_FILTER)
OBJS-$(CONFIG_TMIDEQUALIZER_FILTER) += vf_tmidequalizer.o OBJS-$(CONFIG_TMIDEQUALIZER_FILTER) += vf_tmidequalizer.o
OBJS-$(CONFIG_TMIX_FILTER) += vf_mix.o framesync.o OBJS-$(CONFIG_TMIX_FILTER) += vf_mix.o framesync.o
OBJS-$(CONFIG_TONEMAP_FILTER) += vf_tonemap.o OBJS-$(CONFIG_TONEMAP_FILTER) += vf_tonemap.o
@@ -65,14 +65,14 @@ Index: FFmpeg/libavfilter/allfilters.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/allfilters.c --- FFmpeg.orig/libavfilter/allfilters.c
+++ FFmpeg/libavfilter/allfilters.c +++ FFmpeg/libavfilter/allfilters.c
@@ -494,6 +494,7 @@ extern const AVFilter ff_vf_tmedian; @@ -509,6 +509,7 @@ extern const FFFilter ff_vf_tmedian;
extern const AVFilter ff_vf_tmidequalizer; extern const FFFilter ff_vf_tmidequalizer;
extern const AVFilter ff_vf_tmix; extern const FFFilter ff_vf_tmix;
extern const AVFilter ff_vf_tonemap; extern const FFFilter ff_vf_tonemap;
+extern const AVFilter ff_vf_tonemap_cuda; +extern const FFFilter ff_vf_tonemap_cuda;
extern const AVFilter ff_vf_tonemap_opencl; extern const FFFilter ff_vf_tonemap_opencl;
extern const AVFilter ff_vf_tonemap_vaapi; extern const FFFilter ff_vf_tonemap_vaapi;
extern const AVFilter ff_vf_tpad; extern const FFFilter ff_vf_tpad;
Index: FFmpeg/libavfilter/colorspace.c Index: FFmpeg/libavfilter/colorspace.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/colorspace.c --- FFmpeg.orig/libavfilter/colorspace.c
@@ -128,7 +128,7 @@ Index: FFmpeg/libavfilter/colorspace.c
+ return peak; + return peak;
+} +}
+ +
+void ff_map_dovi_metadata(struct DoviMetadata *out, const AVDOVIMetadata *data) +void ff_map_dovi_metadata(struct FFDOVIMetadataRemap *out, const AVDOVIMetadata *data)
+{ +{
+ int c, i, j, k; + int c, i, j, k;
+ const AVDOVIRpuDataHeader *header; + const AVDOVIRpuDataHeader *header;
@@ -152,7 +152,7 @@ Index: FFmpeg/libavfilter/colorspace.c
+ } + }
+ for (c = 0; c < 3; c++) { + for (c = 0; c < 3; c++) {
+ const AVDOVIReshapingCurve *csrc = &mapping->curves[c]; + const AVDOVIReshapingCurve *csrc = &mapping->curves[c];
+ struct ReshapeData *cdst = &out->comp[c]; + struct FFDOVIReshapeData *cdst = &out->comp[c];
+ cdst->num_pivots = csrc->num_pivots; + cdst->num_pivots = csrc->num_pivots;
+ for (i = 0; i < csrc->num_pivots; i++) { + for (i = 0; i < csrc->num_pivots; i++) {
+ const float scale = 1.0f / ((1 << header->bl_bit_depth) - 1); + const float scale = 1.0f / ((1 << header->bl_bit_depth) - 1);
@@ -183,7 +183,7 @@ Index: FFmpeg/libavfilter/colorspace.c
+} +}
+ +
+// linearizer for PQ/ST2084 +// linearizer for PQ/ST2084
+float eotf_st2084_common(float x) +float ff_eotf_st2084_common(float x)
+{ +{
+ float xpow = powf(FFMAX(x, 0.0f), 1.0f / ST2084_M2); + float xpow = powf(FFMAX(x, 0.0f), 1.0f / ST2084_M2);
+ float num = FFMAX(xpow - ST2084_C1, 0.0f); + float num = FFMAX(xpow - ST2084_C1, 0.0f);
@@ -192,13 +192,13 @@ Index: FFmpeg/libavfilter/colorspace.c
+ return x; + return x;
+} +}
+ +
+float eotf_st2084(float x, float ref_white) +float ff_eotf_st2084(float x, float ref_white)
+{ +{
+ return eotf_st2084_common(x) * ST2084_MAX_LUMINANCE / ref_white; + return ff_eotf_st2084_common(x) * ST2084_MAX_LUMINANCE / ref_white;
+} +}
+ +
+// delinearizer for PQ/ST2084 +// delinearizer for PQ/ST2084
+float inverse_eotf_st2084_common(float x) +float ff_inverse_eotf_st2084_common(float x)
+{ +{
+ float xpow = powf(FFMAX(x, 0.0f), ST2084_M1); + float xpow = powf(FFMAX(x, 0.0f), ST2084_M1);
+#if 0 +#if 0
@@ -214,28 +214,28 @@ Index: FFmpeg/libavfilter/colorspace.c
+#endif +#endif
+} +}
+ +
+float inverse_eotf_st2084(float x, float ref_white) +float ff_inverse_eotf_st2084(float x, float ref_white)
+{ +{
+ x *= ref_white / ST2084_MAX_LUMINANCE; + x *= ref_white / ST2084_MAX_LUMINANCE;
+ return inverse_eotf_st2084_common(x); + return ff_inverse_eotf_st2084_common(x);
+} +}
+ +
+float ootf_1_2(float x) { +float ff_ootf_1_2(float x) {
+ return x > 0.0f ? powf(x, 1.2f) : x; + return x > 0.0f ? powf(x, 1.2f) : x;
+} +}
+ +
+float inverse_ootf_1_2(float x) { +float ff_inverse_ootf_1_2(float x) {
+ return x > 0.0f ? powf(x, 1.0f / 1.2f) : x; + return x > 0.0f ? powf(x, 1.0f / 1.2f) : x;
+} +}
+ +
+float oetf_arib_b67(float x) { +float ff_oetf_arib_b67(float x) {
+ x = FFMAX(x, 0.0f); + x = FFMAX(x, 0.0f);
+ return x <= (1.0f / 12.0f) + return x <= (1.0f / 12.0f)
+ ? sqrtf(3.0f * x) + ? sqrtf(3.0f * x)
+ : (ARIB_B67_A * logf(12.0f * x - ARIB_B67_B) + ARIB_B67_C); + : (ARIB_B67_A * logf(12.0f * x - ARIB_B67_B) + ARIB_B67_C);
+} +}
+ +
+float inverse_oetf_arib_b67(float x) { +float ff_inverse_oetf_arib_b67(float x) {
+ x = FFMAX(x, 0.0f); + x = FFMAX(x, 0.0f);
+ return x <= 0.5f + return x <= 0.5f
+ ? (x * x) * (1.0f / 3.0f) + ? (x * x) * (1.0f / 3.0f)
@@ -243,17 +243,17 @@ Index: FFmpeg/libavfilter/colorspace.c
+} +}
+ +
+// linearizer for HLG/ARIB-B67 +// linearizer for HLG/ARIB-B67
+float eotf_arib_b67(float x) { +float ff_eotf_arib_b67(float x) {
+ return ootf_1_2(inverse_oetf_arib_b67(x)) * (12.0f / REFERENCE_WHITE_HLG); + return ff_ootf_1_2(ff_inverse_oetf_arib_b67(x)) * (12.0f / REFERENCE_WHITE_HLG);
+} +}
+ +
+// delinearizer for HLG/ARIB-B67 +// delinearizer for HLG/ARIB-B67
+float inverse_eotf_arib_b67(float x) { +float ff_inverse_eotf_arib_b67(float x) {
+ return oetf_arib_b67(inverse_ootf_1_2(x / (12.0f / REFERENCE_WHITE_HLG))); + return ff_oetf_arib_b67(ff_inverse_ootf_1_2(x / (12.0f / REFERENCE_WHITE_HLG)));
+} +}
+ +
+// delinearizer for BT709, BT2020-10 +// delinearizer for BT709, BT2020-10
+float inverse_eotf_bt1886(float x) { +float ff_inverse_eotf_bt1886(float x) {
+ return x > 0.0f ? powf(x, 1.0f / 2.4f) : 0.0f; + return x > 0.0f ? powf(x, 1.0f / 2.4f) : 0.0f;
+} +}
Index: FFmpeg/libavfilter/colorspace.h Index: FFmpeg/libavfilter/colorspace.h
@@ -305,13 +305,13 @@ Index: FFmpeg/libavfilter/colorspace.h
+#define QUANTIZATION_OFFSET(n) ((double)(1 << (16 - (n))) / ((1 << 16) - 1)) +#define QUANTIZATION_OFFSET(n) ((double)(1 << (16 - (n))) / ((1 << 16) - 1))
+ +
+// Parsed metadata from the Dolby Vision RPU +// Parsed metadata from the Dolby Vision RPU
+struct DoviMetadata { +struct FFDOVIMetadataRemap {
+ float nonlinear_offset[3]; // input offset ("ycc_to_rgb_offset") + float nonlinear_offset[3]; // input offset ("ycc_to_rgb_offset")
+ double nonlinear[3][3]; // before PQ, also called "ycc_to_rgb" + double nonlinear[3][3]; // before PQ, also called "ycc_to_rgb"
+ double linear[3][3]; // after PQ, also called "rgb_to_lms" + double linear[3][3]; // after PQ, also called "rgb_to_lms"
+ +
+ // Reshape data, grouped by component + // Reshape data, grouped by component
+ struct ReshapeData { + struct FFDOVIReshapeData {
+ uint8_t num_pivots; + uint8_t num_pivots;
+ float pivots[9]; // normalized to [0.0, 1.0] based on BL bit depth + float pivots[9]; // normalized to [0.0, 1.0] based on BL bit depth
+ uint8_t method[8]; // 0 = polynomial, 1 = MMR + uint8_t method[8]; // 0 = polynomial, 1 = MMR
@@ -333,19 +333,19 @@ Index: FFmpeg/libavfilter/colorspace.h
void ff_update_hdr_metadata(AVFrame *in, double peak); void ff_update_hdr_metadata(AVFrame *in, double peak);
+double ff_determine_dovi_signal_peak(const AVDOVIMetadata *data, int l0_only); +double ff_determine_dovi_signal_peak(const AVDOVIMetadata *data, int l0_only);
+void ff_map_dovi_metadata(struct DoviMetadata *out, const AVDOVIMetadata *data); +void ff_map_dovi_metadata(struct FFDOVIMetadataRemap *out, const AVDOVIMetadata *data);
+ +
+float eotf_st2084_common(float x); +float ff_eotf_st2084_common(float x);
+float eotf_st2084(float x, float ref_white); +float ff_eotf_st2084(float x, float ref_white);
+float inverse_eotf_st2084_common(float x); +float ff_inverse_eotf_st2084_common(float x);
+float inverse_eotf_st2084(float x, float ref_white); +float ff_inverse_eotf_st2084(float x, float ref_white);
+float ootf_1_2(float x); +float ff_ootf_1_2(float x);
+float inverse_ootf_1_2(float x); +float ff_inverse_ootf_1_2(float x);
+float oetf_arib_b67(float x); +float ff_oetf_arib_b67(float x);
+float inverse_oetf_arib_b67(float x); +float ff_inverse_oetf_arib_b67(float x);
+float eotf_arib_b67(float x); +float ff_eotf_arib_b67(float x);
+float inverse_eotf_arib_b67(float x); +float ff_inverse_eotf_arib_b67(float x);
+float inverse_eotf_bt1886(float x); +float ff_inverse_eotf_bt1886(float x);
+ +
#endif #endif
Index: FFmpeg/libavfilter/cuda/colorspace_common.h Index: FFmpeg/libavfilter/cuda/colorspace_common.h
@@ -500,12 +500,12 @@ Index: FFmpeg/libavfilter/cuda/colorspace_common.h
+ +
+// linearizer for HLG/ARIB-B67 +// linearizer for HLG/ARIB-B67
+static __inline__ __device__ float eotf_arib_b67(float x) { +static __inline__ __device__ float eotf_arib_b67(float x) {
+ return ootf_1_2(inverse_oetf_arib_b67(x)) * (12.0f / REFERENCE_WHITE_ALT); + return ootf_1_2(inverse_oetf_arib_b67(x)) * (12.0f / REFERENCE_WHITE_HLG);
+} +}
+ +
+// delinearizer for HLG/ARIB-B67 +// delinearizer for HLG/ARIB-B67
+static __inline__ __device__ float inverse_eotf_arib_b67(float x) { +static __inline__ __device__ float inverse_eotf_arib_b67(float x) {
+ return oetf_arib_b67(inverse_ootf_1_2(x / (12.0f / REFERENCE_WHITE_ALT))); + return oetf_arib_b67(inverse_ootf_1_2(x / (12.0f / REFERENCE_WHITE_HLG)));
+} +}
+ +
+// delinearizer for BT709, BT2020-10 +// delinearizer for BT709, BT2020-10
@@ -698,7 +698,7 @@ Index: FFmpeg/libavfilter/cuda/host_util.c
=================================================================== ===================================================================
--- /dev/null --- /dev/null
+++ FFmpeg/libavfilter/cuda/host_util.c +++ FFmpeg/libavfilter/cuda/host_util.c
@@ -0,0 +1,77 @@ @@ -0,0 +1,74 @@
+/* +/*
+ * This file is part of FFmpeg. + * This file is part of FFmpeg.
+ * + *
@@ -725,7 +725,7 @@ Index: FFmpeg/libavfilter/cuda/host_util.c
+#define CHECK_CU(x) FF_CUDA_CHECK_DL(ctx, cu, x) +#define CHECK_CU(x) FF_CUDA_CHECK_DL(ctx, cu, x)
+#define DEPTH_BYTES(depth) (((depth) + 7) / 8) +#define DEPTH_BYTES(depth) (((depth) + 7) / 8)
+ +
+int ff_make_cuda_frame(AVFilterContext *ctx, CudaFunctions *cu, int make_cuTex, +int ff_make_cuda_frame(AVFilterContext *ctx, CudaFunctions *cu, int make_cu_tex,
+ FFCUDAFrame *dst, const AVFrame *src, const AVPixFmtDescriptor *src_desc) + FFCUDAFrame *dst, const AVFrame *src, const AVPixFmtDescriptor *src_desc)
+{ +{
+ int i, ret = 0; + int i, ret = 0;
@@ -738,26 +738,23 @@ Index: FFmpeg/libavfilter/cuda/host_util.c
+ dst->tex[i] = 0; + dst->tex[i] = 0;
+ } + }
+ +
+ for (i = 0; make_cuTex && (i < dst->planes); i++) { + for (i = 0; make_cu_tex && (i < dst->planes); i++) {
+#ifndef CU_TRSF_NORMALIZED_COORDINATES
+ #define CU_TRSF_NORMALIZED_COORDINATES 2
+#endif
+ CUDA_TEXTURE_DESC tex_desc = { + CUDA_TEXTURE_DESC tex_desc = {
+ .addressMode = { CU_TR_ADDRESS_MODE_CLAMP, CU_TR_ADDRESS_MODE_CLAMP }, + .addressMode = { CU_TR_ADDRESS_MODE_CLAMP, CU_TR_ADDRESS_MODE_CLAMP },
+ .filterMode = i == 0 ? CU_TR_FILTER_MODE_POINT : CU_TR_FILTER_MODE_LINEAR, + .filterMode = i ? CU_TR_FILTER_MODE_LINEAR : CU_TR_FILTER_MODE_POINT,
+ .flags = i == 0 ? 0 : CU_TRSF_NORMALIZED_COORDINATES, + .flags = i ? 2 /* CU_TRSF_NORMALIZED_COORDINATES */ : 0
+ }; + };
+ +
+ CUDA_RESOURCE_DESC res_desc = { + CUDA_RESOURCE_DESC res_desc = {
+ .resType = CU_RESOURCE_TYPE_PITCH2D, + .resType = CU_RESOURCE_TYPE_PITCH2D,
+ .res.pitch2D.format = DEPTH_BYTES(src_desc->comp[i].depth) == 1 ? + .res.pitch2D.format = DEPTH_BYTES(src_desc->comp[i].depth) == 1 ?
+ CU_AD_FORMAT_UNSIGNED_INT8 : + CU_AD_FORMAT_UNSIGNED_INT8 :
+ CU_AD_FORMAT_UNSIGNED_INT16, + CU_AD_FORMAT_UNSIGNED_INT16,
+ .res.pitch2D.numChannels = i == 0 ? 1 : (dst->planes == 2 ? 2 : 1), + .res.pitch2D.numChannels = i ? (dst->planes == 2 ? 2 : 1) : 1,
+ .res.pitch2D.width = i == 0 ? src->width : AV_CEIL_RSHIFT(src->width, src_desc->log2_chroma_w), + .res.pitch2D.width = AV_CEIL_RSHIFT(src->width, i ? src_desc->log2_chroma_w : 0),
+ .res.pitch2D.height = i == 0 ? src->height : AV_CEIL_RSHIFT(src->height, src_desc->log2_chroma_h), + .res.pitch2D.height = AV_CEIL_RSHIFT(src->height, i ? src_desc->log2_chroma_h : 0),
+ .res.pitch2D.pitchInBytes = src->linesize[i], + .res.pitch2D.pitchInBytes = src->linesize[i],
+ .res.pitch2D.devPtr = (CUdeviceptr)src->data[i], + .res.pitch2D.devPtr = (CUdeviceptr)src->data[i]
+ }; + };
+ +
+ if ((ret = CHECK_CU(cu->cuTexObjectCreate(&dst->tex[i], &res_desc, &tex_desc, NULL))) < 0) + if ((ret = CHECK_CU(cu->cuTexObjectCreate(&dst->tex[i], &res_desc, &tex_desc, NULL))) < 0)
@@ -807,7 +804,7 @@ Index: FFmpeg/libavfilter/cuda/host_util.h
+#include "libavfilter/avfilter.h" +#include "libavfilter/avfilter.h"
+#include "shared.h" +#include "shared.h"
+ +
+int ff_make_cuda_frame(AVFilterContext *ctx, CudaFunctions *cu, int make_cuTex, +int ff_make_cuda_frame(AVFilterContext *ctx, CudaFunctions *cu, int make_cu_tex,
+ FFCUDAFrame *dst, const AVFrame *src, const AVPixFmtDescriptor *src_desc); + FFCUDAFrame *dst, const AVFrame *src, const AVPixFmtDescriptor *src_desc);
+ +
+#endif /* AVFILTER_CUDA_HOST_UTIL_H */ +#endif /* AVFILTER_CUDA_HOST_UTIL_H */
@@ -815,29 +812,29 @@ Index: FFmpeg/libavfilter/cuda/load_helper.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/cuda/load_helper.c --- FFmpeg.orig/libavfilter/cuda/load_helper.c
+++ FFmpeg/libavfilter/cuda/load_helper.c +++ FFmpeg/libavfilter/cuda/load_helper.c
@@ -32,11 +32,9 @@ @@ -31,11 +31,9 @@
#define CHECK_CU(x) FF_CUDA_CHECK_DL(avctx, cu, x) #define CHECK_CU(x) FF_CUDA_CHECK_DL(avctx, cu, x)
-int ff_cuda_load_module(void *avctx, AVCUDADeviceContext *hwctx, CUmodule *cu_module, -int ff_cuda_load_module(void *avctx, AVCUDADeviceContext *hwctx, CUmodule *cu_module,
- const unsigned char *data, const unsigned int length) - const unsigned char *data, const unsigned int length)
+static int decompress_cuda_ptx(void *avctx, uint8_t **data_out, uint64_t *length_out, +static int decompress_cuda_ptx(void *avctx, uint8_t **data_out, size_t *length_out,
+ const unsigned char *data, const unsigned int length) + const unsigned char *data, const unsigned int length)
{ {
- CudaFunctions *cu = hwctx->internal->cuda_dl; - CudaFunctions *cu = hwctx->internal->cuda_dl;
- -
#if CONFIG_PTX_COMPRESSION #if CONFIG_SHADER_COMPRESSION
z_stream stream = { 0 }; uint8_t *out;
uint8_t *buf, *tmp; size_t out_len;
@@ -88,10 +86,51 @@ int ff_cuda_load_module(void *avctx, AVC @@ -44,10 +42,51 @@ int ff_cuda_load_module(void *avctx, AVC
if (ret < 0)
return ret;
inflateEnd(&stream); - ret = CHECK_CU(cu->cuModuleLoadData(cu_module, out));
- av_free(out);
- ret = CHECK_CU(cu->cuModuleLoadData(cu_module, buf));
- av_free(buf);
- return ret; - return ret;
+ *data_out = buf; + *data_out = out;
+ *length_out = stream.total_out; + *length_out = out_len;
#else #else
- return CHECK_CU(cu->cuModuleLoadData(cu_module, data)); - return CHECK_CU(cu->cuModuleLoadData(cu_module, data));
+ *data_out = NULL; + *data_out = NULL;
@@ -851,7 +848,7 @@ Index: FFmpeg/libavfilter/cuda/load_helper.c
+{ +{
+ CudaFunctions *cu = hwctx->internal->cuda_dl; + CudaFunctions *cu = hwctx->internal->cuda_dl;
+ uint8_t *data_out = NULL; + uint8_t *data_out = NULL;
+ uint64_t length_out = 0; + size_t length_out = 0;
+ int ret; + int ret;
+ +
+ if ((ret = decompress_cuda_ptx(avctx, &data_out, &length_out, data, length)) < 0) + if ((ret = decompress_cuda_ptx(avctx, &data_out, &length_out, data, length)) < 0)
@@ -870,7 +867,7 @@ Index: FFmpeg/libavfilter/cuda/load_helper.c
+{ +{
+ CudaFunctions *cu = hwctx->internal->cuda_dl; + CudaFunctions *cu = hwctx->internal->cuda_dl;
+ uint8_t *data_out = NULL; + uint8_t *data_out = NULL;
+ uint64_t length_out = 0; + size_t length_out = 0;
+ int ret; + int ret;
+ +
+ if ((ret = decompress_cuda_ptx(avctx, &data_out, &length_out, data, length)) < 0) + if ((ret = decompress_cuda_ptx(avctx, &data_out, &length_out, data, length)) < 0)
@@ -1173,7 +1170,7 @@ Index: FFmpeg/libavfilter/cuda/tonemap.cu
=================================================================== ===================================================================
--- /dev/null --- /dev/null
+++ FFmpeg/libavfilter/cuda/tonemap.cu +++ FFmpeg/libavfilter/cuda/tonemap.cu
@@ -0,0 +1,661 @@ @@ -0,0 +1,663 @@
+/* +/*
+ * This file is part of FFmpeg. + * This file is part of FFmpeg.
+ * + *
@@ -1200,6 +1197,7 @@ Index: FFmpeg/libavfilter/cuda/tonemap.cu
+extern __constant__ const enum TonemapAlgorithm tonemap_func; +extern __constant__ const enum TonemapAlgorithm tonemap_func;
+extern __constant__ const float tone_param; +extern __constant__ const float tone_param;
+extern __constant__ const float desat_param; +extern __constant__ const float desat_param;
+extern __constant__ const int lut_size;
+extern __constant__ const int enable_dither; +extern __constant__ const int enable_dither;
+extern __constant__ const float dither_size; +extern __constant__ const float dither_size;
+extern __constant__ const float dither_quantization; +extern __constant__ const float dither_quantization;
@@ -1210,11 +1208,10 @@ Index: FFmpeg/libavfilter/cuda/tonemap.cu
+#define dot3(a, b) ((a).z * (b).z + ((a).y * (b).y + (a).x * (b).x)) +#define dot3(a, b) ((a).z * (b).z + ((a).y * (b).y + (a).x * (b).x))
+#define dot4(a, b) ((a).w * (b).w + ((a).z * (b).z + ((a).y * (b).y + (a).x * (b).x))) +#define dot4(a, b) ((a).w * (b).w + ((a).z * (b).z + ((a).y * (b).y + (a).x * (b).x)))
+ +
+#define LUT_SIZE 65 +template <typename T, typename S>
+
+static __inline__ __device__ +static __inline__ __device__
+float3 clamp3(const float3 a, const float min_val, const float max_val) { +T clamp3(const T a, const S min_val, const S max_val) {
+ float3 result; + T result;
+ result.x = clamp(a.x, min_val, max_val); + result.x = clamp(a.x, min_val, max_val);
+ result.y = clamp(a.y, min_val, max_val); + result.y = clamp(a.y, min_val, max_val);
+ result.z = clamp(a.z, min_val, max_val); + result.z = clamp(a.z, min_val, max_val);
@@ -1573,56 +1570,58 @@ Index: FFmpeg/libavfilter/cuda/tonemap.cu
+ float4 lut_val; + float4 lut_val;
+ color = clamp3(color, 0.0f, 1.0f); + color = clamp3(color, 0.0f, 1.0f);
+ +
+ // Scale the color to the LUT grid. + // Scale the color to the LUT grid
+ float3 pos = color * (float)(LUT_SIZE - 1); + float3 pos = color * (float)(lut_size - 1);
+ +
+ // Get the integer base indices in the LUT. + // Get the integer base indices in the LUT
+ int3 base = make_int3((int)floorf(pos.x), (int)floorf(pos.y), (int)floorf(pos.z)); + int3 base = clamp3(make_int3((int)floorf(pos.x),
+ // Compute the fractional part within the cell. + (int)floorf(pos.y),
+ (int)floorf(pos.z)), 0, lut_size - 2);
+
+ // Compute the fractional part within the cell
+ float3 f = pos - make_float3((float)base.x, (float)base.y, (float)base.z); + float3 f = pos - make_float3((float)base.x, (float)base.y, (float)base.z);
+ +
+ // Compute the base linear index. + // Sort the fraction offsets, so that we always have f_max>=f_mid>=f_min
+ unsigned base_idx = base.x + base.y * LUT_SIZE + base.z * LUT_SIZE * LUT_SIZE; + float f_max = max(f.x, max(f.y, f.z));
+ float f_min = min(f.x, min(f.y, f.z));
+ float f_mid = f.x + f.y + f.z - f_max - f_min;
+ +
+ // Sort the fraction offsets, so that we always have a>=b>=c + // Compute the base linear index
+ float a = max(f.x, max(f.y, f.z)); + unsigned base_idx = base.x + base.y * lut_size + base.z * lut_size * lut_size;
+ float c = min(f.x, min(f.y, f.z)); + unsigned last_idx = base_idx + 1 + lut_size + lut_size * lut_size;
+ float b = f.x + f.y + f.z - a - c;
+
+#define LUT_IDX_MAX (LUT_SIZE * LUT_SIZE * LUT_SIZE - 1)
+ +
+ // The initial and the last corner values of current cube will always be fetched + // The initial and the last corner values of current cube will always be fetched
+ lut_val = lut[min(base_idx, (unsigned)LUT_IDX_MAX)]; + lut_val = lut[base_idx];
+ float3 c000 = make_float3(lut_val.x, lut_val.y, lut_val.z); + float3 c000 = make_float3(lut_val.x, lut_val.y, lut_val.z);
+ lut_val = lut[min(base_idx + 1 + LUT_SIZE + LUT_SIZE * LUT_SIZE, (unsigned)LUT_IDX_MAX)]; + lut_val = lut[last_idx];
+ float3 c111 = make_float3(lut_val.x, lut_val.y, lut_val.z); + float3 c111 = make_float3(lut_val.x, lut_val.y, lut_val.z);
+ +
+ // Select the index for vertices of the tetrahedron. + // Select the index for vertices of the tetrahedron
+ unsigned idx100 = base_idx + 1; + unsigned idx100 = base_idx + 1;
+ unsigned idx010 = base_idx + LUT_SIZE; + unsigned idx010 = base_idx + lut_size;
+ unsigned idx110 = base_idx + 1 + LUT_SIZE; + unsigned idx110 = base_idx + 1 + lut_size;
+ unsigned idx001 = base_idx + LUT_SIZE * LUT_SIZE; + unsigned idx001 = base_idx + lut_size * lut_size;
+ unsigned idx101 = base_idx + 1 + LUT_SIZE * LUT_SIZE; + unsigned idx101 = base_idx + 1 + lut_size * lut_size;
+ unsigned idx011 = base_idx + LUT_SIZE + LUT_SIZE * LUT_SIZE; + unsigned idx011 = base_idx + lut_size + lut_size * lut_size;
+ +
+ // Although we have a and c as max and min value, we cannot use them in the + // Although we have a and c as max and min value, we cannot use them in the
+ // following selection as float equality comparison is not accurate on GPU. + // following selection as float equality comparison is not accurate on GPU
+ unsigned idx0 = select(select(idx001, idx010, (f.y >= f.z && f.y >= f.x)), idx100, (f.x >= f.y && f.x >= f.z)); + unsigned y_max = f.y >= f.z && f.y >= f.x;
+ unsigned idx1 = select(select(idx110, idx101, (f.y <= f.z && f.y <= f.x)), idx011, (f.x <= f.y && f.x <= f.z)); + unsigned x_max = f.x >= f.y && f.x >= f.z;
+ unsigned idx0 = select(select(idx001, idx010, y_max), idx100, x_max);
+ unsigned y_min = f.y <= f.z && f.y <= f.x;
+ unsigned x_min = f.x <= f.y && f.x <= f.z;
+ unsigned idx1 = select(select(idx110, idx101, y_min), idx011, x_min);
+ +
+ // Fetch LUT value with determined tetrahedron + // Fetch LUT value with determined tetrahedron
+ lut_val = lut[min(idx0, (unsigned)LUT_IDX_MAX)]; + lut_val = lut[idx0];
+ float3 c0 = make_float3(lut_val.x, lut_val.y, lut_val.z); + float3 c0 = make_float3(lut_val.x, lut_val.y, lut_val.z);
+ lut_val = lut[min(idx1, (unsigned)LUT_IDX_MAX)]; + lut_val = lut[idx1];
+ float3 c1 = make_float3(lut_val.x, lut_val.y, lut_val.z); + float3 c1 = make_float3(lut_val.x, lut_val.y, lut_val.z);
+ +
+ float3 ca = c0 - c000; + return clamp3(c000 + f_max * (c0 - c000)
+ float3 cb = c1 - c0; + + f_mid * (c1 - c0)
+ float3 cc = c111 - c1; + + f_min * (c111 - c1), 0.0f, 1.0f);
+
+ float3 result = c000 + a * ca + b * cb + c * cc;
+
+ return clamp3(result, 0.0f, 1.0f);
+} +}
+ +
+extern "C" { +extern "C" {
@@ -1775,16 +1774,16 @@ Index: FFmpeg/libavfilter/cuda/tonemap.cu
+ int skip_tonemap, + int skip_tonemap,
+ int dovi_reshape) + int dovi_reshape)
+{ +{
+ const int total_entries = LUT_SIZE * LUT_SIZE * LUT_SIZE; + const int total_entries = lut_size * lut_size * lut_size;
+ int idx = blockIdx.x * blockDim.x + threadIdx.x; + int idx = blockIdx.x * blockDim.x + threadIdx.x;
+ if (idx >= total_entries) return; + if (idx >= total_entries) return;
+ int z = idx / (LUT_SIZE * LUT_SIZE); + int z = idx / (lut_size * lut_size);
+ int rem = idx - (z * LUT_SIZE * LUT_SIZE); + int rem = idx - (z * lut_size * lut_size);
+ int y = rem / LUT_SIZE; + int y = rem / lut_size;
+ int x = rem % LUT_SIZE; + int x = rem % lut_size;
+ float fx = (float)x / (LUT_SIZE - 1); + float fx = (float)x / (lut_size - 1);
+ float fy = (float)y / (LUT_SIZE - 1); + float fy = (float)y / (lut_size - 1);
+ float fz = (float)z / (LUT_SIZE - 1); + float fz = (float)z / (lut_size - 1);
+ float3 c = make_float3(fx, fy, fz); + float3 c = make_float3(fx, fy, fz);
+ if (tonemap_mode == TONEMAP_MODE_ITP) { + if (tonemap_mode == TONEMAP_MODE_ITP) {
+ c = dovi_reshape + c = dovi_reshape
@@ -1994,7 +1993,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_cuda.c
=================================================================== ===================================================================
--- /dev/null --- /dev/null
+++ FFmpeg/libavfilter/vf_tonemap_cuda.c +++ FFmpeg/libavfilter/vf_tonemap_cuda.c
@@ -0,0 +1,1271 @@ @@ -0,0 +1,1281 @@
+/* +/*
+ * This file is part of FFmpeg. + * This file is part of FFmpeg.
+ * + *
@@ -2095,7 +2094,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_cuda.c
+ CUfunction cu_func_dovi; + CUfunction cu_func_dovi;
+ CUfunction cu_func_dovi_pq; + CUfunction cu_func_dovi_pq;
+ +
+#define LUT_SIZE (65 * 65 * 65) +#define LUT_SIZE 65
+ CUdeviceptr lut_buffer; + CUdeviceptr lut_buffer;
+ +
+ CUdeviceptr dither_buffer; + CUdeviceptr dither_buffer;
@@ -2110,11 +2109,13 @@ Index: FFmpeg/libavfilter/vf_tonemap_cuda.c
+#define coeffs_sz coeffs_cnt*sizeof(float) +#define coeffs_sz coeffs_cnt*sizeof(float)
+#define mmr_sz mmr_cnt*sizeof(float) +#define mmr_sz mmr_cnt*sizeof(float)
+ CUdeviceptr dovi_buffer; + CUdeviceptr dovi_buffer;
+ struct DoviMetadata *dovi; + struct FFDOVIMetadataRemap *dovi;
+ float *dovi_pbuf; + float *dovi_pbuf;
+ +
+ enum TonemapAlgorithm tonemap; + /* enum TonemapAlgorithm */
+ enum TonemapMode tonemap_mode; + int tonemap;
+ /* enum TonemapMode */
+ int tonemap_mode;
+ int apply_dovi; + int apply_dovi;
+ int tradeoff; + int tradeoff;
+ int init_with_dovi; + int init_with_dovi;
@@ -2257,30 +2258,28 @@ Index: FFmpeg/libavfilter/vf_tonemap_cuda.c
+ .srcPitch = ff_fruit_dither_size * sizeof(ff_fruit_dither_matrix[0]), + .srcPitch = ff_fruit_dither_size * sizeof(ff_fruit_dither_matrix[0]),
+ .dstPitch = ff_fruit_dither_size * sizeof(ff_fruit_dither_matrix[0]), + .dstPitch = ff_fruit_dither_size * sizeof(ff_fruit_dither_matrix[0]),
+ .WidthInBytes = ff_fruit_dither_size * sizeof(ff_fruit_dither_matrix[0]), + .WidthInBytes = ff_fruit_dither_size * sizeof(ff_fruit_dither_matrix[0]),
+ .Height = ff_fruit_dither_size, + .Height = ff_fruit_dither_size
+ }; + };
+ +
+#ifndef CU_TRSF_NORMALIZED_COORDINATES
+ #define CU_TRSF_NORMALIZED_COORDINATES 2
+#endif
+ CUDA_TEXTURE_DESC tex_desc = { + CUDA_TEXTURE_DESC tex_desc = {
+ .addressMode = { CU_TR_ADDRESS_MODE_WRAP, + .addressMode = { CU_TR_ADDRESS_MODE_WRAP,
+ CU_TR_ADDRESS_MODE_WRAP }, + CU_TR_ADDRESS_MODE_WRAP },
+ .filterMode = CU_TR_FILTER_MODE_POINT, + .filterMode = CU_TR_FILTER_MODE_POINT,
+ .flags = CU_TRSF_NORMALIZED_COORDINATES, + .flags = 2 /* CU_TRSF_NORMALIZED_COORDINATES */
+ }; + };
+ +
+ CUDA_RESOURCE_DESC res_desc = { + CUDA_RESOURCE_DESC res_desc = {
+ .resType = CU_RESOURCE_TYPE_PITCH2D, + .resType = CU_RESOURCE_TYPE_PITCH2D,
+ .res.pitch2D.format = CU_AD_FORMAT_UNSIGNED_INT16, + .res.pitch2D.format = CU_AD_FORMAT_UNSIGNED_INT16,
+ .res.pitch2D.numChannels = 1, + .res.pitch2D.numChannels = 1,
+ .res.pitch2D.width = ff_fruit_dither_size, + .res.pitch2D.width = ff_fruit_dither_size,
+ .res.pitch2D.height = ff_fruit_dither_size, + .res.pitch2D.height = ff_fruit_dither_size,
+ .res.pitch2D.pitchInBytes = ff_fruit_dither_size * sizeof(ff_fruit_dither_matrix[0]), + .res.pitch2D.pitchInBytes = ff_fruit_dither_size * sizeof(ff_fruit_dither_matrix[0]),
+ .res.pitch2D.devPtr = 0, + .res.pitch2D.devPtr = 0
+ }; + };
+ +
+ av_assert0(sizeof(ff_fruit_dither_matrix) == sizeof(ff_fruit_dither_matrix[0]) * ff_fruit_dither_size * ff_fruit_dither_size); + av_assert0(sizeof(ff_fruit_dither_matrix) ==
+ sizeof(ff_fruit_dither_matrix[0]) * ff_fruit_dither_size * ff_fruit_dither_size);
+ +
+ if ((ret = CHECK_CU(cu->cuCtxPushCurrent(cuda_ctx))) < 0) + if ((ret = CHECK_CU(cu->cuCtxPushCurrent(cuda_ctx))) < 0)
+ return ret; + return ret;
@@ -2444,7 +2443,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_cuda.c
+ for (c = 0; c < 3; c++) { + for (c = 0; c < 3; c++) {
+ int has_poly = 0, has_mmr = 0, mmr_single = 1; + int has_poly = 0, has_mmr = 0, mmr_single = 1;
+ int mmr_idx = 0, min_order = 3, max_order = 1; + int mmr_idx = 0, min_order = 3, max_order = 1;
+ const struct ReshapeData *comp = &s->dovi->comp[c]; + const struct FFDOVIReshapeData *comp = &s->dovi->comp[c];
+ if (!comp->num_pivots) + if (!comp->num_pivots)
+ continue; + continue;
+ av_assert0(comp->num_pivots >= 2 && comp->num_pivots <= 9); + av_assert0(comp->num_pivots >= 2 && comp->num_pivots <= 9);
@@ -2714,6 +2713,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_cuda.c
+ CONSTANT(".u32 chroma_loc_src = %i", (int)s->in_chroma_loc); + CONSTANT(".u32 chroma_loc_src = %i", (int)s->in_chroma_loc);
+ CONSTANT(".u32 chroma_loc_dst = %i", (int)s->out_chroma_loc); + CONSTANT(".u32 chroma_loc_dst = %i", (int)s->out_chroma_loc);
+ CONSTANT(".u32 tonemap_func = %i", (int)s->tonemap); + CONSTANT(".u32 tonemap_func = %i", (int)s->tonemap);
+ CONSTANT(".u32 lut_size = %i", (int)(LUT_SIZE));
+ CONSTANT(".u32 enable_dither = %i", (int)(s->in_desc->comp[0].depth > s->out_desc->comp[0].depth)); + CONSTANT(".u32 enable_dither = %i", (int)(s->in_desc->comp[0].depth > s->out_desc->comp[0].depth));
+ CONSTANT(".f32 dither_size = %.1f", (float)ff_fruit_dither_size); + CONSTANT(".f32 dither_size = %.1f", (float)ff_fruit_dither_size);
+ CONSTANT(".f32 dither_quantization = %.1f", (float)((1 << s->out_desc->comp[0].depth) - 1)); + CONSTANT(".f32 dither_quantization = %.1f", (float)((1 << s->out_desc->comp[0].depth) - 1));
@@ -2783,6 +2783,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_cuda.c
+ if (ret < 0) + if (ret < 0)
+ goto fail; + goto fail;
+ +
+ s->cu_func_build_lut = NULL;
+ s->cu_func_tm = NULL; + s->cu_func_tm = NULL;
+ s->cu_func_dovi = NULL; + s->cu_func_dovi = NULL;
+ s->cu_func_dovi_pq = NULL; + s->cu_func_dovi_pq = NULL;
@@ -2813,7 +2814,8 @@ Index: FFmpeg/libavfilter/vf_tonemap_cuda.c
+ +
+ if (s->tradeoff == 1) { + if (s->tradeoff == 1) {
+ const size_t lut_size = LUT_SIZE; + const size_t lut_size = LUT_SIZE;
+ const size_t lut_buffer_size = lut_size * sizeof(float) * 4; + const size_t lut_size_3d = lut_size * lut_size * lut_size;
+ const size_t lut_buffer_size = lut_size_3d * sizeof(float) * 4;
+ float peak = (float)s->src_peak; + float peak = (float)s->src_peak;
+ float dst_peak = (float)s->dst_peak; + float dst_peak = (float)s->dst_peak;
+ int skip_tonemap = s->out_trc == AVCOL_TRC_SMPTE2084; + int skip_tonemap = s->out_trc == AVCOL_TRC_SMPTE2084;
@@ -2840,7 +2842,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_cuda.c
+ void *args[] = { &s->lut_buffer, &peak, &dst_peak, &s->tonemap_mode, &skip_tonemap, &dovi_reshape }; + void *args[] = { &s->lut_buffer, &peak, &dst_peak, &s->tonemap_mode, &skip_tonemap, &dovi_reshape };
+ +
+ ret = CHECK_CU(cu->cuLaunchKernel(s->cu_func_build_lut, + ret = CHECK_CU(cu->cuLaunchKernel(s->cu_func_build_lut,
+ lut_size, 1, 1, + lut_size_3d, 1, 1,
+ 1, 1, 1, 0, s->hwctx->stream, args, NULL)); + 1, 1, 1, 0, s->hwctx->stream, args, NULL));
+ if (ret < 0) + if (ret < 0)
+ goto fail2; + goto fail2;
@@ -2917,6 +2919,13 @@ Index: FFmpeg/libavfilter/vf_tonemap_cuda.c
+ +
+ outlink->sample_aspect_ratio = inlink->sample_aspect_ratio; + outlink->sample_aspect_ratio = inlink->sample_aspect_ratio;
+ +
+ if (s->trc != AVCOL_TRC_SMPTE2084) {
+ av_frame_side_data_remove(&outlink->side_data, &outlink->nb_side_data,
+ AV_FRAME_DATA_MASTERING_DISPLAY_METADATA);
+ av_frame_side_data_remove(&outlink->side_data, &outlink->nb_side_data,
+ AV_FRAME_DATA_CONTENT_LIGHT_LEVEL);
+ }
+
+ return 0; + return 0;
+} +}
+ +
@@ -3049,7 +3058,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_cuda.c
+ const AVDOVIRpuDataHeader *rpu = av_dovi_get_header(metadata); + const AVDOVIRpuDataHeader *rpu = av_dovi_get_header(metadata);
+ // only map dovi rpus that don't require an EL + // only map dovi rpus that don't require an EL
+ if (rpu->disable_residual_flag) { + if (rpu->disable_residual_flag) {
+ struct DoviMetadata *dovi = av_malloc(sizeof(*dovi)); + struct FFDOVIMetadataRemap *dovi = av_malloc(sizeof(*dovi));
+ s->dovi = dovi; + s->dovi = dovi;
+ if (!s->dovi) + if (!s->dovi)
+ goto fail; + goto fail;
@@ -3248,16 +3257,16 @@ Index: FFmpeg/libavfilter/vf_tonemap_cuda.c
+ }, + },
+}; +};
+ +
+const AVFilter ff_vf_tonemap_cuda = { +const FFFilter ff_vf_tonemap_cuda = {
+ .name = "tonemap_cuda", + .p.name = "tonemap_cuda",
+ .description = NULL_IF_CONFIG_SMALL("GPU accelerated HDR to SDR tonemapping"), + .p.description = NULL_IF_CONFIG_SMALL("GPU accelerated HDR to SDR tonemapping"),
+ +
+ .preinit = preinit, + .preinit = preinit,
+ .init = init, + .init = init,
+ .uninit = uninit, + .uninit = uninit,
+ +
+ .priv_size = sizeof(TonemapCUDAContext), + .priv_size = sizeof(TonemapCUDAContext),
+ .priv_class = &tonemap_cuda_class, + .p.priv_class = &tonemap_cuda_class,
+ +
+ FILTER_INPUTS(tonemap_cuda_inputs), + FILTER_INPUTS(tonemap_cuda_inputs),
+ FILTER_OUTPUTS(tonemap_cuda_outputs), + FILTER_OUTPUTS(tonemap_cuda_outputs),
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
@@ -501,7 +501,7 @@ Index: FFmpeg/libavfilter/opencl/tonemap.cl
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/opencl/tonemap.cl --- FFmpeg.orig/libavfilter/opencl/tonemap.cl
+++ FFmpeg/libavfilter/opencl/tonemap.cl +++ FFmpeg/libavfilter/opencl/tonemap.cl
@@ -16,54 +16,89 @@ @@ -16,54 +16,88 @@
* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
*/ */
@@ -528,7 +528,6 @@ Index: FFmpeg/libavfilter/opencl/tonemap.cl
+#endif +#endif
+ +
+#define FLOAT_EPS 1e-6f +#define FLOAT_EPS 1e-6f
+#define LUT_SIZE 65
+ +
extern float3 lrgb2yuv(float3); extern float3 lrgb2yuv(float3);
extern float lrgb2y(float3); extern float lrgb2y(float3);
@@ -612,7 +611,7 @@ Index: FFmpeg/libavfilter/opencl/tonemap.cl
float j = tone_param; float j = tone_param;
float a, b; float a, b;
@@ -71,202 +106,1016 @@ float mobius(float s, float peak) { @@ -71,202 +105,1040 @@ float mobius(float s, float peak) {
return s; return s;
a = -j * j * (peak - 1.0f) / (j * j - 2.0f * j + peak); a = -j * j * (peak - 1.0f) / (j * j - 2.0f * j + peak);
@@ -1565,59 +1564,83 @@ Index: FFmpeg/libavfilter/opencl/tonemap.cl
+#endif +#endif
+} +}
+ +
+float3 apply_lut3d(__global const float3 *restrict lut, float3 color) +#undef lut3d_read_t
+#ifdef LUT_PERF_IMAGE3D
+ #define lut3d_read_t __read_only image3d_t
+#else
+ #define lut3d_read_t __global const float4 *restrict
+#endif
+
+float3 apply_lut3d(lut3d_read_t lut, float3 color)
+{ +{
+ color = clamp(color, 0.0f, 1.0f); + color = clamp(color, 0.0f, 1.0f);
+ +
+ // Scale the color to the LUT grid. + // Scale the color to the LUT grid
+ float3 pos = color * (float)(LUT_SIZE - 1); + float3 pos = color * (float)(LUT_SIZE - 1);
+ +
+ // Get the integer base indices in the LUT. + // Get the integer base indices in the LUT
+ int3 base = convert_int3(floor(pos)); + int3 base = clamp(convert_int3(floor(pos)), 0, LUT_SIZE - 2);
+ // Compute the fractional part within the cell. +
+ // Compute the fractional part within the cell
+ float3 f = pos - convert_float3(base); + float3 f = pos - convert_float3(base);
+ +
+ // Compute the base linear index. + // Sort the fraction offsets, so that we always have f_max>=f_mid>=f_min
+ uint base_idx = base.x + base.y * LUT_SIZE + base.z * LUT_SIZE * LUT_SIZE; + float f_max = fmax(f.x, fmax(f.y, f.z));
+ + float f_min = fmin(f.x, fmin(f.y, f.z));
+ // Sort the fraction offsets, so that we always have a>=b>=c + float f_mid = f.x + f.y + f.z - f_max - f_min;
+ float a = fmax(f.x, fmax(f.y, f.z));
+ float c = fmin(f.x, fmin(f.y, f.z));
+ float b = f.x + f.y + f.z - a - c;
+
+#define LUT_IDX_MAX (LUT_SIZE * LUT_SIZE * LUT_SIZE - 1)
+ +
+ // The initial and the last corner values of current cube will always be fetched + // The initial and the last corner values of current cube will always be fetched
+ float3 c000 = lut[min(base_idx, (uint)LUT_IDX_MAX)]; +#ifdef LUT_PERF_IMAGE3D
+ float3 c111 = lut[min(base_idx + 1 + LUT_SIZE + LUT_SIZE * LUT_SIZE, (uint)LUT_IDX_MAX)]; + float3 c000 = read_imagef(lut, n_sampler, (int4)(base + 0, 0)).xyz;
+ float3 c111 = read_imagef(lut, n_sampler, (int4)(base + 1, 0)).xyz;
+#else
+ // Compute the base linear index
+ uint base_idx = base.x + base.y * LUT_SIZE + base.z * LUT_SIZE * LUT_SIZE;
+ uint last_idx = base_idx + 1 + LUT_SIZE + LUT_SIZE * LUT_SIZE;
+ float3 c000 = lut[base_idx].xyz;
+ float3 c111 = lut[last_idx].xyz;
+#endif
+ +
+ // Select the index for vertices of the tetrahedron. + // Select the index for vertices of the tetrahedron:
+ // Although we have f_max and f_min values, we cannot use them in the
+ // following selection as float equality comparison is not accurate on GPU
+#ifdef LUT_PERF_IMAGE3D
+ int3 y_max = (int3)(-(f.y >= f.z && f.y >= f.x));
+ int3 x_max = (int3)(-(f.x >= f.y && f.x >= f.z));
+ int3 d0 = select(select((int3)(0, 0, 1), (int3)(0, 1, 0), y_max), (int3)(1, 0, 0), x_max);
+ int3 y_min = (int3)(-(f.y <= f.z && f.y <= f.x));
+ int3 x_min = (int3)(-(f.x <= f.y && f.x <= f.z));
+ int3 d1 = select(select((int3)(1, 1, 0), (int3)(1, 0, 1), y_min), (int3)(0, 1, 1), x_min);
+#else
+ uint idx100 = base_idx + 1; + uint idx100 = base_idx + 1;
+ uint idx010 = base_idx + LUT_SIZE; + uint idx010 = base_idx + LUT_SIZE;
+ uint idx110 = base_idx + 1 + LUT_SIZE; + uint idx110 = base_idx + 1 + LUT_SIZE;
+ uint idx001 = base_idx + LUT_SIZE * LUT_SIZE; + uint idx001 = base_idx + LUT_SIZE * LUT_SIZE;
+ uint idx101 = base_idx + 1 + LUT_SIZE * LUT_SIZE; + uint idx101 = base_idx + 1 + LUT_SIZE * LUT_SIZE;
+ uint idx011 = base_idx + LUT_SIZE + LUT_SIZE * LUT_SIZE; + uint idx011 = base_idx + LUT_SIZE + LUT_SIZE * LUT_SIZE;
+ + uint y_max = f.y >= f.z && f.y >= f.x;
+ // Although we have a and c as max and min value, we cannot use them in the + uint x_max = f.x >= f.y && f.x >= f.z;
+ // following selection as float equality comparison is not accurate on GPU. + uint idx0 = select(select(idx001, idx010, y_max), idx100, x_max);
+ uint idx0 = select(select(idx001, idx010, (f.y >= f.z && f.y >= f.x)), idx100, (f.x >= f.y && f.x >= f.z)); + uint y_min = f.y <= f.z && f.y <= f.x;
+ uint idx1 = select(select(idx110, idx101, (f.y <= f.z && f.y <= f.x)), idx011, (f.x <= f.y && f.x <= f.z)); + uint x_min = f.x <= f.y && f.x <= f.z;
+ uint idx1 = select(select(idx110, idx101, y_min), idx011, x_min);
+#endif
+ +
+ // Fetch LUT value with determined tetrahedron + // Fetch LUT value with determined tetrahedron
+ float3 c0 = lut[min(idx0, (uint)LUT_IDX_MAX)]; +#ifdef LUT_PERF_IMAGE3D
+ float3 c1 = lut[min(idx1, (uint)LUT_IDX_MAX)]; + float3 c0 = read_imagef(lut, n_sampler, (int4)(base + d0, 0)).xyz;
+ float3 c1 = read_imagef(lut, n_sampler, (int4)(base + d1, 0)).xyz;
+#else
+ float3 c0 = lut[idx0].xyz;
+ float3 c1 = lut[idx1].xyz;
+#endif
+ +
+ float3 ca = c0 - c000; + return clamp(c000 + f_max * (c0 - c000)
+ float3 cb = c1 - c0; + + f_mid * (c1 - c0)
+ float3 cc = c111 - c1; + + f_min * (c111 - c1), 0.0f, 1.0f);
+
+ float3 result = c000 + a * ca + b * cb + c * cc;
+
+ return clamp(result, 0.0f, 1.0f);
+} +}
+ +
+__kernel void tonemap_lut(__global const float3 *restrict lut, +__kernel void tonemap_lut( lut3d_read_t lut,
+ __write_only image2d_t dst1, + __write_only image2d_t dst1,
+ __read_only image2d_t src1, + __read_only image2d_t src1,
+ __write_only image2d_t dst2, + __write_only image2d_t dst2,
@@ -1753,7 +1776,7 @@ Index: FFmpeg/libavfilter/opencl/tonemap.cl
+#endif +#endif
+} +}
+ +
+__kernel void build_lut(__global float3 *lut, float peak) +__kernel void build_lut(__global float4 *lut, float peak)
+{ +{
+ const int total_entries = LUT_SIZE * LUT_SIZE * LUT_SIZE; + const int total_entries = LUT_SIZE * LUT_SIZE * LUT_SIZE;
+ int idx = get_global_id(0); + int idx = get_global_id(0);
@@ -1790,7 +1813,7 @@ Index: FFmpeg/libavfilter/opencl/tonemap.cl
+ c = clamp(c, 0.0f, 1.0f); + c = clamp(c, 0.0f, 1.0f);
+#endif +#endif
+ c = lrgb2yuv(c); + c = lrgb2yuv(c);
+ lut[idx] = clamp(c, 0.0f, 1.0f); + lut[idx] = clamp((float4)(c, 0.0f), 0.0f, 1.0f);
} }
Index: FFmpeg/libavfilter/vf_tonemap_opencl.c Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
=================================================================== ===================================================================
@@ -1802,13 +1825,14 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
* This file is part of FFmpeg. * This file is part of FFmpeg.
* *
* FFmpeg is free software; you can redistribute it and/or * FFmpeg is free software; you can redistribute it and/or
@@ -15,27 +15,49 @@ @@ -15,27 +15,50 @@
* License along with FFmpeg; if not, write to the Free Software * License along with FFmpeg; if not, write to the Free Software
* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
*/ */
+ +
#include <float.h> #include <float.h>
+#define CL_USE_DEPRECATED_OPENCL_1_2_APIS
+#ifdef __APPLE__ +#ifdef __APPLE__
+#include <OpenCL/cl_ext.h> +#include <OpenCL/cl_ext.h>
+#else +#else
@@ -1858,7 +1882,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
enum TonemapAlgorithm { enum TonemapAlgorithm {
TONEMAP_NONE, TONEMAP_NONE,
@@ -45,7 +67,17 @@ enum TonemapAlgorithm { @@ -45,7 +68,17 @@ enum TonemapAlgorithm {
TONEMAP_REINHARD, TONEMAP_REINHARD,
TONEMAP_HABLE, TONEMAP_HABLE,
TONEMAP_MOBIUS, TONEMAP_MOBIUS,
@@ -1877,7 +1901,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
}; };
typedef struct TonemapOpenCLContext { typedef struct TonemapOpenCLContext {
@@ -56,23 +88,48 @@ typedef struct TonemapOpenCLContext { @@ -56,24 +89,52 @@ typedef struct TonemapOpenCLContext {
enum AVColorPrimaries primaries, primaries_in, primaries_out; enum AVColorPrimaries primaries, primaries_in, primaries_out;
enum AVColorRange range, range_in, range_out; enum AVColorRange range, range_in, range_out;
enum AVChromaLocation chroma_loc; enum AVChromaLocation chroma_loc;
@@ -1893,14 +1917,16 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
+#define pivots_sz pivots_cnt*sizeof(cl_float) +#define pivots_sz pivots_cnt*sizeof(cl_float)
+#define coeffs_sz coeffs_cnt*sizeof(cl_float) +#define coeffs_sz coeffs_cnt*sizeof(cl_float)
+#define mmr_sz mmr_cnt*sizeof(cl_float) +#define mmr_sz mmr_cnt*sizeof(cl_float)
+ struct DoviMetadata *dovi; + struct FFDOVIMetadataRemap *dovi;
+ cl_mem dovi_buf; + cl_mem dovi_buf;
+ unsigned dovi_use_fp16; + unsigned dovi_use_fp16;
+ unsigned is_pure_dovi; + unsigned is_pure_dovi;
+#define LUT_SIZE (65 * 65 * 65) +#define LUT_SIZE 65
enum TonemapAlgorithm tonemap; /* enum TonemapAlgorithm */
+ enum TonemapMode tonemap_mode; int tonemap;
+ /* enum TonemapMode */
+ int tonemap_mode;
enum AVPixelFormat format; enum AVPixelFormat format;
+ int apply_dovi; + int apply_dovi;
double peak; double peak;
@@ -1912,25 +1938,27 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
- double target_peak; - double target_peak;
double scene_threshold; double scene_threshold;
+ int tradeoff; + int tradeoff;
+ int use_image3d;
int initialised; int initialised;
+ int init_with_dovi; + int init_with_dovi;
cl_kernel kernel; cl_kernel kernel;
+ cl_kernel lut_generation_kernel; + cl_kernel lut_generation_kernel;
+ cl_mem dither_image; + cl_mem dither_image;
+ cl_mem lut_buffer; + cl_mem lut_buffer;
+ cl_mem lut_image;
cl_command_queue command_queue; cl_command_queue command_queue;
- cl_mem util_mem; - cl_mem util_mem;
} TonemapOpenCLContext; } TonemapOpenCLContext;
static const char *const linearize_funcs[AVCOL_TRC_NB] = { static const char *const linearize_funcs[] = {
- [AVCOL_TRC_SMPTE2084] = "eotf_st2084", - [AVCOL_TRC_SMPTE2084] = "eotf_st2084",
- [AVCOL_TRC_ARIB_STD_B67] = "inverse_oetf_hlg", - [AVCOL_TRC_ARIB_STD_B67] = "inverse_oetf_hlg",
+ [AVCOL_TRC_SMPTE2084] = "eotf_st2084", + [AVCOL_TRC_SMPTE2084] = "eotf_st2084",
+ [AVCOL_TRC_ARIB_STD_B67] = "eotf_arib_b67", + [AVCOL_TRC_ARIB_STD_B67] = "eotf_arib_b67",
}; };
static const char *const delinearize_funcs[AVCOL_TRC_NB] = { static const char *const delinearize_funcs[] = {
@@ -80,7 +137,7 @@ static const char *const delinearize_fun @@ -81,7 +142,7 @@ static const char *const delinearize_fun
[AVCOL_TRC_BT2020_10] = "inverse_eotf_bt1886", [AVCOL_TRC_BT2020_10] = "inverse_eotf_bt1886",
}; };
@@ -1939,7 +1967,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
[TONEMAP_NONE] = "direct", [TONEMAP_NONE] = "direct",
[TONEMAP_LINEAR] = "linear", [TONEMAP_LINEAR] = "linear",
[TONEMAP_GAMMA] = "gamma", [TONEMAP_GAMMA] = "gamma",
@@ -88,6 +145,14 @@ static const char *const tonemap_func[TO @@ -89,6 +150,14 @@ static const char *const tonemap_func[TO
[TONEMAP_REINHARD] = "reinhard", [TONEMAP_REINHARD] = "reinhard",
[TONEMAP_HABLE] = "hable", [TONEMAP_HABLE] = "hable",
[TONEMAP_MOBIUS] = "mobius", [TONEMAP_MOBIUS] = "mobius",
@@ -1954,7 +1982,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
}; };
static int get_rgb2rgb_matrix(enum AVColorPrimaries in, enum AVColorPrimaries out, static int get_rgb2rgb_matrix(enum AVColorPrimaries in, enum AVColorPrimaries out,
@@ -108,90 +173,453 @@ static int get_rgb2rgb_matrix(enum AVCol @@ -109,90 +178,510 @@ static int get_rgb2rgb_matrix(enum AVCol
return 0; return 0;
} }
@@ -1995,7 +2023,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
+ cl_float mmr_packed_dataf[8*6][4] = {0}; + cl_float mmr_packed_dataf[8*6][4] = {0};
+ cl_half coeffs_datah[8][4] = {0}; + cl_half coeffs_datah[8][4] = {0};
+ cl_half mmr_packed_datah[8*6][4] = {0}; + cl_half mmr_packed_datah[8*6][4] = {0};
+ int c, i, j, k, err; + int c, i, j, k, err av_unused;
+ cl_int cle; + cl_int cle;
+ +
+ pbuf = clEnqueueMapBuffer(ctx->command_queue, ctx->dovi_buf, + pbuf = clEnqueueMapBuffer(ctx->command_queue, ctx->dovi_buf,
@@ -2008,7 +2036,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
+ for (c = 0; c < 3; c++) { + for (c = 0; c < 3; c++) {
+ int has_poly = 0, has_mmr = 0, mmr_single = 1; + int has_poly = 0, has_mmr = 0, mmr_single = 1;
+ int mmr_idx = 0, min_order = 3, max_order = 1; + int mmr_idx = 0, min_order = 3, max_order = 1;
+ const struct ReshapeData *comp = &ctx->dovi->comp[c]; + const struct FFDOVIReshapeData *comp = &ctx->dovi->comp[c];
+ if (!comp->num_pivots) + if (!comp->num_pivots)
+ continue; + continue;
+ av_assert0(comp->num_pivots >= 2 && comp->num_pivots <= 9); + av_assert0(comp->num_pivots >= 2 && comp->num_pivots <= 9);
@@ -2150,24 +2178,20 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
TonemapOpenCLContext *ctx = avctx->priv; TonemapOpenCLContext *ctx = avctx->priv;
+ AVBPrint header; + AVBPrint header;
+ const char *opencl_sources[OPENCL_SOURCE_NB]; + const char *opencl_sources[OPENCL_SOURCE_NB];
+ size_t m_origin[3] = {0};
+ size_t m_region[3] = {ff_fruit_dither_size, ff_fruit_dither_size, 1};
+ size_t m_row_pitch = ff_fruit_dither_size * sizeof(ff_fruit_dither_matrix[0]);
int rgb2rgb_passthrough = 1; int rgb2rgb_passthrough = 1;
double rgb2rgb[3][3], rgb2yuv[3][3], yuv2rgb[3][3]; double rgb2rgb[3][3], rgb2yuv[3][3], yuv2rgb[3][3];
const AVLumaCoefficients *luma_src, *luma_dst; const AVLumaCoefficients *luma_src, *luma_dst;
+ cl_event event = NULL; + cl_event event_in = NULL, event_out = NULL;
+ cl_bool device_is_integrated; + cl_mem_flags dovi_buf_flags = CL_MEM_ALLOC_HOST_PTR | CL_MEM_HOST_WRITE_ONLY | CL_MEM_READ_ONLY;
+ cl_uint max_compute_units, device_vendor_id; + cl_uint device_vendor_id;
cl_int cle; cl_int cle;
- int err; - int err;
- AVBPrint header; - AVBPrint header;
- const char *opencl_sources[OPENCL_SOURCE_NB]; - const char *opencl_sources[OPENCL_SOURCE_NB];
+ cl_mem_flags dovi_buf_flags = CL_MEM_ALLOC_HOST_PTR | CL_MEM_HOST_WRITE_ONLY | CL_MEM_READ_ONLY;
+ char *device_vendor = NULL; + char *device_vendor = NULL;
+ char *device_name = NULL; + char *device_name = NULL;
+ char *device_exts = NULL; + char *device_exts = NULL;
+ int is_device_qualcomm = 0; + int is_qcom_proprietary = 0;
+ int i, j, err; + int i, j, err;
- av_bprint_init(&header, 1024, AV_BPRINT_SIZE_AUTOMATIC); - av_bprint_init(&header, 1024, AV_BPRINT_SIZE_AUTOMATIC);
@@ -2214,54 +2238,120 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
ctx->target_peak = 1.0f; ctx->target_peak = 1.0f;
- av_log(ctx, AV_LOG_DEBUG, "tone mapping transfer from %s to %s\n", - av_log(ctx, AV_LOG_DEBUG, "tone mapping transfer from %s to %s\n",
+ +
+ cle = clGetDeviceInfo(ctx->ocf.hwctx->device_id, CL_DEVICE_VENDOR_ID, + cle = clGetDeviceInfo(ctx->ocf.hwctx->device_id,
+ sizeof(cl_uint), &device_vendor_id, + CL_DEVICE_VENDOR_ID,
+ NULL); + sizeof(cl_uint), &device_vendor_id, NULL);
+ CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to check OpenCL " + CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to check OpenCL "
+ "device vendor id %d.\n", cle); + "device vendor id %d.\n", cle);
+ +
+ if (ctx->tradeoff == -1) { + device_exts = check_opencl_device_str(ctx->ocf.hwctx->device_id,
+ ctx->tradeoff = 1; + CL_DEVICE_EXTENSIONS);
+ cle = clGetDeviceInfo(ctx->ocf.hwctx->device_id, CL_DEVICE_HOST_UNIFIED_MEMORY, +
+ sizeof(cl_bool), &device_is_integrated, + ctx->use_image3d = 0;
+ NULL); + if (ctx->tradeoff) {
+ cl_bool device_is_uma = 0;
+ int is_intel = 0, is_arm = 0, is_qcom = 0;
+ int is_tradeoff_auto = ctx->tradeoff == -1;
+
+ cle = clGetDeviceInfo(ctx->ocf.hwctx->device_id,
+ CL_DEVICE_HOST_UNIFIED_MEMORY,
+ sizeof(cl_bool), &device_is_uma, NULL);
+ CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to check if OpenCL " + CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to check if OpenCL "
+ "device is integrated %d.\n", cle); + "device is UMA %d.\n", cle);
+ cle = clGetDeviceInfo(ctx->ocf.hwctx->device_id, CL_DEVICE_MAX_COMPUTE_UNITS, +
+ sizeof(cl_uint), &max_compute_units, + device_vendor = check_opencl_device_str(ctx->ocf.hwctx->device_id,
+ NULL); + CL_DEVICE_VENDOR);
+ CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to check OpenCL " + device_name = check_opencl_device_str(ctx->ocf.hwctx->device_id,
+ "device max compute units %d.\n", cle); + CL_DEVICE_NAME);
+ if (device_vendor_id == 0x8086 && device_is_integrated == CL_TRUE) { +
+ is_intel = device_vendor_id == 0x8086;
+ is_arm = device_vendor_id == 0x13b5 ||
+ (device_vendor && strstr(device_vendor, "ARM")) ||
+ (device_name && strstr(device_name, "Mali"));
+ is_qcom = device_vendor_id == 0x5143 ||
+ device_vendor_id == MKTAG('Q', 'C', 'O', 'M');
+
+ ctx->tradeoff = 1;
+ if (is_intel && device_is_uma && is_tradeoff_auto) {
+ // Use tradeoff on low perf Intel iGPUs
+ cl_uint max_compute_units = 0;
+
+ cle = clGetDeviceInfo(ctx->ocf.hwctx->device_id,
+ CL_DEVICE_MAX_COMPUTE_UNITS,
+ sizeof(cl_uint), &max_compute_units, NULL);
+ CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to check OpenCL "
+ "device max compute units %d.\n", cle);
+
+ if (max_compute_units >= 40) + if (max_compute_units >= 40)
+ ctx->tradeoff = 0; + ctx->tradeoff = 0;
+ if (device_name = check_opencl_device_str(ctx->ocf.hwctx->device_id, CL_DEVICE_NAME)) { + else if (device_name) {
+ const char *excluded_devices[5] = { "Arc", "Iris", "Xe", "770", "750" }; + const char *excluded_devices[5] = { "Arc", "Iris", "Xe", "770", "750" };
+ for (i = 0; i < FF_ARRAY_ELEMS(excluded_devices); i++) { + for (i = 0; i < FF_ARRAY_ELEMS(excluded_devices); i++) {
+ if (strstr(device_name, excluded_devices[i])) { + if (strstr(device_name, excluded_devices[i])) {
+ ctx->tradeoff = 0; + ctx->tradeoff = 0; break;
+ break;
+ } + }
+ } + }
+ av_freep(&device_name);
+ } + }
+ } else if (device_vendor_id == 0x5143 || device_vendor_id == MKTAG('Q', 'C', 'O', 'M')) { + } else if (is_arm && device_is_uma) {
+ // Use tradeoff and check image3d_t support for lut on ARM Mali Valhall+
+ cl_uint nb_formats = 0;
+ cl_image_format *formats = NULL;
+
+ if (!(device_exts && strstr(device_exts, "cl_arm_job_slot_selection"))) {
+ cle = clGetSupportedImageFormats(ctx->ocf.hwctx->context,
+ CL_MEM_READ_ONLY,
+ CL_MEM_OBJECT_IMAGE3D,
+ 0, NULL, &nb_formats);
+ if (cle == CL_SUCCESS && nb_formats > 0) {
+ formats = av_malloc_array(nb_formats, sizeof(*formats));
+ if (!formats) {
+ err = AVERROR(ENOMEM);
+ goto fail;
+ }
+ cle = clGetSupportedImageFormats(ctx->ocf.hwctx->context,
+ CL_MEM_READ_ONLY,
+ CL_MEM_OBJECT_IMAGE3D,
+ nb_formats, formats, NULL);
+ for (i = 0; cle == CL_SUCCESS && i < nb_formats; i++) {
+ if (formats[i].image_channel_order == CL_RGBA &&
+ formats[i].image_channel_data_type == CL_FLOAT) {
+ ctx->use_image3d = 1; break;
+ }
+ }
+ }
+ av_freep(&formats);
+ }
+ if (ctx->use_image3d) {
+ size_t value = 0;
+ cl_device_info params[] = {
+ CL_DEVICE_IMAGE3D_MAX_WIDTH,
+ CL_DEVICE_IMAGE3D_MAX_HEIGHT,
+ CL_DEVICE_IMAGE3D_MAX_DEPTH
+ };
+
+ for (i = 0; i < FF_ARRAY_ELEMS(params); i++) {
+ cle = clGetDeviceInfo(ctx->ocf.hwctx->device_id, params[i],
+ sizeof(value), &value, NULL);
+ if (cle != CL_SUCCESS || value < LUT_SIZE) {
+ ctx->use_image3d = 0; break;
+ }
+ }
+ }
+ if (!ctx->use_image3d)
+ av_log(avctx, AV_LOG_DEBUG,
+ "Disabled image3d for lut due to lack of support.\n");
+ } else if (is_qcom) {
+ // Always use tradeoff on Qualcomm due to inconsistent performance + // Always use tradeoff on Qualcomm due to inconsistent performance
+ ctx->tradeoff = 1; + } else if (is_tradeoff_auto) {
+ } else if (device_is_integrated == CL_TRUE) {
+ device_vendor = check_opencl_device_str(ctx->ocf.hwctx->device_id, CL_DEVICE_VENDOR);
+ device_name = check_opencl_device_str(ctx->ocf.hwctx->device_id, CL_DEVICE_NAME);
+ if (!strstr(device_vendor, "ARM") &&
+ !strstr(device_name, "Mali"))
+ ctx->tradeoff = 0;
+ av_freep(&device_vendor);
+ av_freep(&device_name);
+ } else {
+ ctx->tradeoff = 0; + ctx->tradeoff = 0;
+ } + }
+ +
+ if (!ctx->tradeoff) + if (is_tradeoff_auto && !ctx->tradeoff)
+ av_log(avctx, AV_LOG_DEBUG, "Disabled tradeoffs on high performance device.\n"); + av_log(avctx, AV_LOG_DEBUG,
+ "Disabled tradeoffs on high performance device.\n");
+
+ av_freep(&device_vendor);
+ av_freep(&device_name);
+ } + }
+ +
+ // for low perf device, only do reshaping for pure dovi + // for low perf device, only do reshaping for pure dovi
@@ -2270,17 +2360,19 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
+ ctx->apply_dovi = 0; + ctx->apply_dovi = 0;
+ } + }
+ +
+ device_exts = check_opencl_device_str(ctx->ocf.hwctx->device_id, CL_DEVICE_EXTENSIONS);
+
+ // use FP16 for dovi reshaping only when tradeoff is enabled and it's supported + // use FP16 for dovi reshaping only when tradeoff is enabled and it's supported
+ ctx->dovi_use_fp16 = 0; + ctx->dovi_use_fp16 = 0;
+ if (ctx->tradeoff && ctx->dovi) { + if (ctx->tradeoff && ctx->dovi && device_exts && strstr(device_exts, "cl_khr_fp16")) {
+ if (device_exts && strstr(device_exts, "cl_khr_fp16")) { + ctx->dovi_use_fp16 = 1;
+ ctx->dovi_use_fp16 = 1; + av_log(avctx, AV_LOG_DEBUG, "FP16 is enabled for DOVI reshaping.\n");
+ av_log(avctx, AV_LOG_DEBUG, "FP16 is enabled for DOVI reshaping.\n");
+ }
+ } + }
+ +
+ // zero-copy buffer requires this extension on Intel dGPUs
+ if (device_vendor_id == 0x8086 && device_exts && strstr(device_exts, "cl_intel_mem_force_host_memory"))
+ dovi_buf_flags |= (1 << 20); /* CL_MEM_FORCE_HOST_MEMORY_INTEL */
+
+ av_freep(&device_exts);
+
+ if (device_vendor_id == 0x5143) { + if (device_vendor_id == 0x5143) {
+ // Qualcomm has two device IDs: 0x5143 and 0x4d4f4351 ('Q' | 'C' << 8 | 'O' << 16 | 'M' << 24) + // Qualcomm has two device IDs: 0x5143 and 0x4d4f4351 ('Q' | 'C' << 8 | 'O' << 16 | 'M' << 24)
+ // The former is reported by Qualcomm's official OpenCL driver + // The former is reported by Qualcomm's official OpenCL driver
@@ -2288,21 +2380,10 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
+ // The former has better performance if the kernel is written in a way its compiler handles properly + // The former has better performance if the kernel is written in a way its compiler handles properly
+ // The latter one has more predictable performance and compiler behaves a bit more like other GPU + // The latter one has more predictable performance and compiler behaves a bit more like other GPU
+ // Only use the workaround on Qualcomm native OpenCL driver + // Only use the workaround on Qualcomm native OpenCL driver
+ is_qcom_proprietary = 1;
+ av_log(avctx, AV_LOG_DEBUG, "Qualcomm driver in use, vendor specific workarounds applied.\n"); + av_log(avctx, AV_LOG_DEBUG, "Qualcomm driver in use, vendor specific workarounds applied.\n");
+ is_device_qualcomm = 1;
+ } + }
+ +
+#ifndef CL_MEM_FORCE_HOST_MEMORY_INTEL
+ #define CL_MEM_FORCE_HOST_MEMORY_INTEL (1 << 20)
+#endif
+ // zero-copy buffer requires this extension on Intel dGPUs
+ if (device_vendor_id == 0x8086) {
+ if (device_exts && strstr(device_exts, "cl_intel_mem_force_host_memory"))
+ dovi_buf_flags |= CL_MEM_FORCE_HOST_MEMORY_INTEL;
+ }
+
+ av_freep(&device_exts);
+
+ av_log(ctx, AV_LOG_DEBUG, "Tonemapping transfer from %s to %s\n", + av_log(ctx, AV_LOG_DEBUG, "Tonemapping transfer from %s to %s\n",
av_color_transfer_name(ctx->trc_in), av_color_transfer_name(ctx->trc_in),
av_color_transfer_name(ctx->trc_out)); av_color_transfer_name(ctx->trc_out));
@@ -2319,7 +2400,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
+ av_log(ctx, AV_LOG_DEBUG, "Mapping range from %s to %s\n", + av_log(ctx, AV_LOG_DEBUG, "Mapping range from %s to %s\n",
av_color_range_name(ctx->range_in), av_color_range_name(ctx->range_in),
av_color_range_name(ctx->range_out)); av_color_range_name(ctx->range_out));
- // checking valid value just because of limited implementaion - // checking valid value just because of limited implementation
- // please remove when more functionalities are implemented - // please remove when more functionalities are implemented
+ +
av_assert0(ctx->trc_out == AVCOL_TRC_BT709 || av_assert0(ctx->trc_out == AVCOL_TRC_BT709 ||
@@ -2353,9 +2434,13 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
+ +
+ av_bprint_init(&header, 2048, AV_BPRINT_SIZE_UNLIMITED); + av_bprint_init(&header, 2048, AV_BPRINT_SIZE_UNLIMITED);
+ +
+ if (is_device_qualcomm) + if (is_qcom_proprietary)
+ av_bprintf(&header, "#define IS_QCOM_GPU\n"); + av_bprintf(&header, "#define IS_QCOM_GPU\n");
+ +
+ av_bprintf(&header, "#define LUT_SIZE %d\n", LUT_SIZE);
+ if (ctx->use_image3d)
+ av_bprintf(&header, "#define LUT_PERF_IMAGE3D\n");
+
+ av_bprintf(&header, "__constant float tone_param = %.4ff;\n", + av_bprintf(&header, "__constant float tone_param = %.4ff;\n",
+ ctx->final_param); + ctx->final_param);
+ av_bprintf(&header, "__constant float desat_param = %.4ff;\n", + av_bprintf(&header, "__constant float desat_param = %.4ff;\n",
@@ -2437,7 +2522,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
av_bprintf(&header, "#define chroma_loc %d\n", (int)ctx->chroma_loc); av_bprintf(&header, "#define chroma_loc %d\n", (int)ctx->chroma_loc);
if (rgb2rgb_passthrough) if (rgb2rgb_passthrough)
@@ -199,19 +627,41 @@ static int tonemap_opencl_init(AVFilterC @@ -200,19 +689,44 @@ static int tonemap_opencl_init(AVFilterC
else else
ff_opencl_print_const_matrix_3x3(&header, "rgb2rgb", rgb2rgb); ff_opencl_print_const_matrix_3x3(&header, "rgb2rgb", rgb2rgb);
@@ -2451,6 +2536,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
- ctx->colorspace_in, av_color_space_name(ctx->colorspace_in)); - ctx->colorspace_in, av_color_space_name(ctx->colorspace_in));
- goto fail; - goto fail;
+ if (ctx->dovi) { + if (ctx->dovi) {
+ const size_t buf_sz = 3*(params_sz+pivots_sz+coeffs_sz+mmr_sz) >> !!ctx->dovi_use_fp16;
+ double ycc2rgb_offset[3] = {0}; + double ycc2rgb_offset[3] = {0};
+ double lms2rgb[3][3]; + double lms2rgb[3][3];
+ av_bprintf(&header, "#define DOVI_RESHAPE\n"); + av_bprintf(&header, "#define DOVI_RESHAPE\n");
@@ -2465,6 +2551,8 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
+ ff_matrix_mul_3x3(lms2rgb, dovi_lms2rgb_matrix, ctx->dovi->linear); + ff_matrix_mul_3x3(lms2rgb, dovi_lms2rgb_matrix, ctx->dovi->linear);
+ ff_opencl_print_const_matrix_3x3(&header, "rgb_matrix", ctx->dovi->nonlinear); //ycc2rgb + ff_opencl_print_const_matrix_3x3(&header, "rgb_matrix", ctx->dovi->nonlinear); //ycc2rgb
+ ff_opencl_print_const_matrix_3x3(&header, "lms2rgb_matrix", lms2rgb); //lms2rgb + ff_opencl_print_const_matrix_3x3(&header, "lms2rgb_matrix", lms2rgb); //lms2rgb
+
+ CL_CREATE_BUFFER_FLAGS(ctx, dovi_buf, dovi_buf_flags, buf_sz, NULL);
+ } else { + } else {
+ luma_src = av_csp_luma_coeffs_from_avcsp(ctx->colorspace_in); + luma_src = av_csp_luma_coeffs_from_avcsp(ctx->colorspace_in);
+ if (!luma_src) { + if (!luma_src) {
@@ -2486,7 +2574,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
ctx->colorspace_out, av_color_space_name(ctx->colorspace_out)); ctx->colorspace_out, av_color_space_name(ctx->colorspace_out));
goto fail; goto fail;
} }
@@ -219,24 +669,13 @@ static int tonemap_opencl_init(AVFilterC @@ -220,24 +734,13 @@ static int tonemap_opencl_init(AVFilterC
ff_fill_rgb2yuv_table(luma_dst, rgb2yuv); ff_fill_rgb2yuv_table(luma_dst, rgb2yuv);
ff_opencl_print_const_matrix_3x3(&header, "yuv_matrix", rgb2yuv); ff_opencl_print_const_matrix_3x3(&header, "yuv_matrix", rgb2yuv);
@@ -2516,13 +2604,17 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
av_log(avctx, AV_LOG_DEBUG, "Generated OpenCL header:\n%s\n", header.str); av_log(avctx, AV_LOG_DEBUG, "Generated OpenCL header:\n%s\n", header.str);
opencl_sources[0] = header.str; opencl_sources[0] = header.str;
@@ -254,46 +693,231 @@ static int tonemap_opencl_init(AVFilterC @@ -255,50 +758,291 @@ static int tonemap_opencl_init(AVFilterC
CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to create OpenCL " CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to create OpenCL "
"command queue %d.\n", cle); "command queue %d.\n", cle);
- ctx->kernel = clCreateKernel(ctx->ocf.program, "tonemap", &cle); - ctx->kernel = clCreateKernel(ctx->ocf.program, "tonemap", &cle);
- CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to create kernel %d.\n", cle); - CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to create kernel %d.\n", cle);
+ if (ctx->in_desc->comp[0].depth > ctx->out_desc->comp[0].depth) { + if (ctx->in_desc->comp[0].depth > ctx->out_desc->comp[0].depth) {
+ const size_t m_origin[3] = { 0 };
+ const size_t m_region[3] = { ff_fruit_dither_size, ff_fruit_dither_size, 1 };
+ const size_t m_row_pitch = ff_fruit_dither_size * sizeof(ff_fruit_dither_matrix[0]);
+
+ cl_image_format image_format = { + cl_image_format image_format = {
+ .image_channel_data_type = CL_UNORM_INT16, + .image_channel_data_type = CL_UNORM_INT16,
+ .image_channel_order = CL_R, + .image_channel_order = CL_R,
@@ -2534,7 +2626,8 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
+ .image_row_pitch = 0, + .image_row_pitch = 0,
+ }; + };
+ +
+ av_assert0(sizeof(ff_fruit_dither_matrix) == sizeof(ff_fruit_dither_matrix[0]) * ff_fruit_dither_size * ff_fruit_dither_size); + av_assert0(sizeof(ff_fruit_dither_matrix) ==
+ sizeof(ff_fruit_dither_matrix[0]) * ff_fruit_dither_size * ff_fruit_dither_size);
+ +
+ ctx->dither_image = clCreateImage(ctx->ocf.hwctx->context, CL_MEM_READ_ONLY, + ctx->dither_image = clCreateImage(ctx->ocf.hwctx->context, CL_MEM_READ_ONLY,
+ &image_format, &image_desc, NULL, &cle); + &image_format, &image_desc, NULL, &cle);
@@ -2550,53 +2643,86 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
+ CL_FALSE, m_origin, m_region, + CL_FALSE, m_origin, m_region,
+ m_row_pitch, 0, + m_row_pitch, 0,
+ ff_fruit_dither_matrix, + ff_fruit_dither_matrix,
+ 0, NULL, &event); + 0, NULL, &event_out);
+ CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to enqueue write of dither matrix image: %d.\n", cle); + CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to enqueue write of dither matrix image: %d.\n", cle);
+ +
+ cle = clWaitForEvents(1, &event); + cle = clWaitForEvents(1, &event_out);
+ CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to wait for event completion: %d.\n", cle); + CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to wait for event completion: %d.\n", cle);
+ if (event) { + if (event_out) {
+ clReleaseEvent(event); + clReleaseEvent(event_out);
+ event = NULL; + event_out = NULL;
+ } + }
+ } + }
+
+ if (ctx->tradeoff) {
+ const size_t lut_size = LUT_SIZE;
+ const size_t lut_buffer_size = lut_size * sizeof(cl_float3);
+ float peak = (float)ctx->src_peak;
+
+ ctx->lut_generation_kernel = clCreateKernel(ctx->ocf.program, "build_lut", &cle);
+ CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to create kernel %d.\n", cle);
+
+ CL_CREATE_BUFFER_FLAGS(ctx, lut_buffer, CL_MEM_HOST_NO_ACCESS | CL_MEM_READ_WRITE, lut_buffer_size, NULL);
+ CL_SET_KERNEL_ARG(ctx->lut_generation_kernel, 0, cl_mem, &ctx->lut_buffer);
+ CL_SET_KERNEL_ARG(ctx->lut_generation_kernel, 1, cl_float, &peak);
+
+ cle = clEnqueueNDRangeKernel(ctx->command_queue, ctx->lut_generation_kernel, 1, NULL,
+ &lut_size, NULL,
+ 0, NULL, NULL);
+ CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to enqueue build_lut kernel: %d.\n", cle);
+
+ cle = clFinish(ctx->command_queue);
+ CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to finish command queue: %d.\n", cle);
+
+ ctx->kernel = clCreateKernel(ctx->ocf.program, "tonemap_lut", &cle);
+ CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to create kernel %d.\n", cle);
+ } else {
+ ctx->kernel = clCreateKernel(ctx->ocf.program, "tonemap", &cle);
+ CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to create kernel %d.\n", cle);
+ }
- ctx->util_mem = - ctx->util_mem =
- clCreateBuffer(ctx->ocf.hwctx->context, 0, - clCreateBuffer(ctx->ocf.hwctx->context, 0,
- (2 * DETECTION_FRAMES + 7) * sizeof(unsigned), - (2 * DETECTION_FRAMES + 7) * sizeof(unsigned),
- NULL, &cle); - NULL, &cle);
- CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to create util buffer: %d.\n", cle); - CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to create util buffer: %d.\n", cle);
+ if (ctx->dovi) { + if (ctx->tradeoff) {
+ const size_t buf_sz = 3*(params_sz+pivots_sz+coeffs_sz+mmr_sz) >> !!ctx->dovi_use_fp16; + const size_t lut_size = LUT_SIZE;
+ CL_CREATE_BUFFER_FLAGS(ctx, dovi_buf, dovi_buf_flags, buf_sz, NULL); + const size_t lut_size_3d = lut_size * lut_size * lut_size;
+ const size_t lut_buffer_size = lut_size_3d * sizeof(cl_float4);
+ const float peak = (float)ctx->src_peak;
+ const size_t m_origin[3] = { 0 };
+ const size_t m_region[3] = { lut_size, lut_size, lut_size };
+ cl_mem_flags mem_flags = CL_MEM_HOST_NO_ACCESS;
+
+ cl_image_format image_format = {
+ .image_channel_order = CL_RGBA,
+ .image_channel_data_type = CL_FLOAT,
+ };
+ cl_image_desc image_desc = {
+ .image_type = CL_MEM_OBJECT_IMAGE3D,
+ .image_width = lut_size,
+ .image_height = lut_size,
+ .image_depth = lut_size,
+ .image_row_pitch = 0,
+ .image_slice_pitch = 0,
+ };
+
+ ctx->lut_generation_kernel = clCreateKernel(ctx->ocf.program, "build_lut", &cle);
+ CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to create kernel %d.\n", cle);
+
+ CL_CREATE_BUFFER_FLAGS(ctx, lut_buffer, CL_MEM_READ_WRITE | mem_flags, lut_buffer_size, NULL);
+
+ CL_SET_KERNEL_ARG(ctx->lut_generation_kernel, 0, cl_mem, &ctx->lut_buffer);
+ CL_SET_KERNEL_ARG(ctx->lut_generation_kernel, 1, cl_float, &peak);
+ cle = clEnqueueNDRangeKernel(ctx->command_queue, ctx->lut_generation_kernel, 1, NULL,
+ &lut_size_3d, NULL,
+ 0, NULL, ctx->use_image3d ? &event_in : &event_out);
+ CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to enqueue build_lut kernel: %d.\n", cle);
+
+ if (ctx->use_image3d) {
+ ctx->lut_image = clCreateImage(ctx->ocf.hwctx->context, CL_MEM_READ_ONLY | mem_flags,
+ &image_format, &image_desc, NULL, &err);
+ if (!ctx->lut_image) {
+ av_log(avctx, AV_LOG_ERROR, "Failed to create image for "
+ "lut image: %d.\n", cle);
+ err = AVERROR(EIO);
+ goto fail;
+ }
+ cle = clEnqueueCopyBufferToImage(ctx->command_queue,
+ ctx->lut_buffer, ctx->lut_image,
+ 0, m_origin, m_region,
+ 1, &event_in, &event_out);
+ CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to enqueue copy of lut buffer to image: %d.\n", cle);
+ }
+
+ cle = clWaitForEvents(1, &event_out);
+ CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to wait for event completion: %d.\n", cle);
+ if (event_in) {
+ clReleaseEvent(event_in);
+ event_in = NULL;
+ }
+ if (event_out) {
+ clReleaseEvent(event_out);
+ event_out = NULL;
+ }
+ } + }
+
+ ctx->kernel = clCreateKernel(ctx->ocf.program, ctx->tradeoff ? "tonemap_lut" : "tonemap", &cle);
+ CL_FAIL_ON_ERROR(AVERROR(EIO), "Failed to create kernel %d.\n", cle);
ctx->initialised = 1; ctx->initialised = 1;
return 0; return 0;
@@ -2610,8 +2736,10 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
+ av_freep(&device_vendor); + av_freep(&device_vendor);
+ av_freep(&device_name); + av_freep(&device_name);
+ av_freep(&device_exts); + av_freep(&device_exts);
+ if (event) + if (event_in)
+ clReleaseEvent(event); + clReleaseEvent(event_in);
+ if (event_out)
+ clReleaseEvent(event_out);
if (ctx->kernel) if (ctx->kernel)
clReleaseKernel(ctx->kernel); clReleaseKernel(ctx->kernel);
+ if (ctx->lut_generation_kernel) + if (ctx->lut_generation_kernel)
@@ -2620,6 +2748,8 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
+ clReleaseMemObject(ctx->dither_image); + clReleaseMemObject(ctx->dither_image);
+ if (ctx->lut_buffer) + if (ctx->lut_buffer)
+ clReleaseMemObject(ctx->lut_buffer); + clReleaseMemObject(ctx->lut_buffer);
+ if (ctx->lut_image)
+ clReleaseMemObject(ctx->lut_image);
+ if (ctx->dovi_buf) + if (ctx->dovi_buf)
+ clReleaseMemObject(ctx->dovi_buf); + clReleaseMemObject(ctx->dovi_buf);
+ if (ctx->command_queue) + if (ctx->command_queue)
@@ -2686,6 +2816,13 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
+ "lut buffer: %d.\n", cle); + "lut buffer: %d.\n", cle);
+ } + }
+ +
+ if (ctx->lut_image) {
+ cle = clReleaseMemObject(ctx->lut_image);
+ if (cle != CL_SUCCESS)
+ av_log(avctx, AV_LOG_ERROR, "Failed to release "
+ "lut image: %d.\n", cle);
+ }
+
+ if (ctx->command_queue) { + if (ctx->command_queue) {
+ cle = clReleaseCommandQueue(ctx->command_queue); + cle = clReleaseCommandQueue(ctx->command_queue);
+ if (cle != CL_SUCCESS) + if (cle != CL_SUCCESS)
@@ -2751,13 +2888,14 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
+ av_log(ctx, AV_LOG_ERROR, "Unsupported output format: %s\n", + av_log(ctx, AV_LOG_ERROR, "Unsupported output format: %s\n",
+ av_get_pix_fmt_name(out_format)); + av_get_pix_fmt_name(out_format));
+ return AVERROR(ENOSYS); + return AVERROR(ENOSYS);
} + }
+ if (in_desc->comp[0].depth != 10 && in_desc->comp[0].depth != 16) { + if (in_desc->comp[0].depth != 10 && in_desc->comp[0].depth != 16) {
+ av_log(ctx, AV_LOG_ERROR, "Unsupported input format depth: %d\n", + av_log(ctx, AV_LOG_ERROR, "Unsupported input format depth: %d\n",
+ in_desc->comp[0].depth); + in_desc->comp[0].depth);
+ return AVERROR(ENOSYS); + return AVERROR(ENOSYS);
+ } }
+
- s->ocf.output_format = s->format == AV_PIX_FMT_NONE ? AV_PIX_FMT_NV12 : s->format;
+ ctx->in_fmt = in_format; + ctx->in_fmt = in_format;
+ ctx->out_fmt = out_format; + ctx->out_fmt = out_format;
+ ctx->in_desc = in_desc; + ctx->in_desc = in_desc;
@@ -2765,12 +2903,22 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
+ ctx->in_planes = av_pix_fmt_count_planes(in_format); + ctx->in_planes = av_pix_fmt_count_planes(in_format);
+ ctx->out_planes = av_pix_fmt_count_planes(out_format); + ctx->out_planes = av_pix_fmt_count_planes(out_format);
+ ctx->ocf.output_format = out_format; + ctx->ocf.output_format = out_format;
+
- s->ocf.output_format = s->format == AV_PIX_FMT_NONE ? AV_PIX_FMT_NV12 : s->format;
ret = ff_opencl_filter_config_output(outlink); ret = ff_opencl_filter_config_output(outlink);
if (ret < 0) if (ret < 0)
return ret; return ret;
@@ -308,13 +932,49 @@ static int launch_kernel(AVFilterContext
+ if (ctx->trc != AVCOL_TRC_SMPTE2084) {
+ av_frame_side_data_remove(&outlink->side_data, &outlink->nb_side_data,
+ AV_FRAME_DATA_MASTERING_DISPLAY_METADATA);
+ av_frame_side_data_remove(&outlink->side_data, &outlink->nb_side_data,
+ AV_FRAME_DATA_CONTENT_LIGHT_LEVEL);
+ }
+
return 0;
}
@@ -309,13 +1053,50 @@ static int launch_kernel(AVFilterContext
size_t global_work[2]; size_t global_work[2];
size_t local_work[2]; size_t local_work[2];
cl_int cle; cl_int cle;
@@ -2785,6 +2933,11 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
+ err = AVERROR(EIO); + err = AVERROR(EIO);
+ goto fail; + goto fail;
+ } + }
+
+ if (ctx->in_planes > 2 && !input->data[2]) {
+ err = AVERROR(EIO);
+ goto fail;
+ }
- CL_SET_KERNEL_ARG(kernel, 0, cl_mem, &output->data[0]); - CL_SET_KERNEL_ARG(kernel, 0, cl_mem, &output->data[0]);
- CL_SET_KERNEL_ARG(kernel, 1, cl_mem, &input->data[0]); - CL_SET_KERNEL_ARG(kernel, 1, cl_mem, &input->data[0]);
@@ -2792,14 +2945,10 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
- CL_SET_KERNEL_ARG(kernel, 3, cl_mem, &input->data[1]); - CL_SET_KERNEL_ARG(kernel, 3, cl_mem, &input->data[1]);
- CL_SET_KERNEL_ARG(kernel, 4, cl_mem, &ctx->util_mem); - CL_SET_KERNEL_ARG(kernel, 4, cl_mem, &ctx->util_mem);
- CL_SET_KERNEL_ARG(kernel, 5, cl_float, &peak); - CL_SET_KERNEL_ARG(kernel, 5, cl_float, &peak);
+ if (ctx->in_planes > 2 && !input->data[2]) {
+ err = AVERROR(EIO);
+ goto fail;
+ }
+
+ idx_arg = 0; + idx_arg = 0;
+ if (ctx->tradeoff) { + if (ctx->tradeoff) {
+ CL_SET_KERNEL_ARG(kernel, idx_arg++, cl_mem, &ctx->lut_buffer); + CL_SET_KERNEL_ARG(kernel, idx_arg++, cl_mem,
+ ctx->use_image3d ? &ctx->lut_image : &ctx->lut_buffer);
+ } + }
+ CL_SET_KERNEL_ARG(kernel, idx_arg++, cl_mem, &output->data[0]); + CL_SET_KERNEL_ARG(kernel, idx_arg++, cl_mem, &output->data[0]);
+ CL_SET_KERNEL_ARG(kernel, idx_arg++, cl_mem, &input->data[0]); + CL_SET_KERNEL_ARG(kernel, idx_arg++, cl_mem, &input->data[0]);
@@ -2826,7 +2975,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
local_work[0] = 16; local_work[0] = 16;
local_work[1] = 16; local_work[1] = 16;
@@ -338,12 +998,10 @@ static int tonemap_opencl_filter_frame(A @@ -339,12 +1120,10 @@ static int tonemap_opencl_filter_frame(A
AVFilterContext *avctx = inlink->dst; AVFilterContext *avctx = inlink->dst;
AVFilterLink *outlink = avctx->outputs[0]; AVFilterLink *outlink = avctx->outputs[0];
TonemapOpenCLContext *ctx = avctx->priv; TonemapOpenCLContext *ctx = avctx->priv;
@@ -2840,7 +2989,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
av_log(ctx, AV_LOG_DEBUG, "Filter input: %s, %ux%u (%"PRId64").\n", av_log(ctx, AV_LOG_DEBUG, "Filter input: %s, %ux%u (%"PRId64").\n",
av_get_pix_fmt_name(input->format), av_get_pix_fmt_name(input->format),
@@ -351,7 +1009,6 @@ static int tonemap_opencl_filter_frame(A @@ -352,7 +1131,6 @@ static int tonemap_opencl_filter_frame(A
if (!input->hw_frames_ctx) if (!input->hw_frames_ctx)
return AVERROR(EINVAL); return AVERROR(EINVAL);
@@ -2848,7 +2997,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
output = ff_get_video_buffer(outlink, outlink->w, outlink->h); output = ff_get_video_buffer(outlink, outlink->w, outlink->h);
if (!output) { if (!output) {
@@ -363,17 +1020,65 @@ static int tonemap_opencl_filter_frame(A @@ -364,17 +1142,65 @@ static int tonemap_opencl_filter_frame(A
if (err < 0) if (err < 0)
goto fail; goto fail;
@@ -2890,7 +3039,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
+ const AVDOVIRpuDataHeader *rpu = av_dovi_get_header(metadata); + const AVDOVIRpuDataHeader *rpu = av_dovi_get_header(metadata);
+ // only map dovi rpus that don't require an EL + // only map dovi rpus that don't require an EL
+ if (rpu->disable_residual_flag) { + if (rpu->disable_residual_flag) {
+ struct DoviMetadata *dovi = av_malloc(sizeof(*dovi)); + struct FFDOVIMetadataRemap *dovi = av_malloc(sizeof(*dovi));
+ ctx->dovi = dovi; + ctx->dovi = dovi;
+ if (!ctx->dovi) + if (!ctx->dovi)
+ goto fail; + goto fail;
@@ -2920,7 +3069,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
ctx->trc_in = input->color_trc; ctx->trc_in = input->color_trc;
ctx->trc_out = output->color_trc; ctx->trc_out = output->color_trc;
@@ -385,72 +1090,50 @@ static int tonemap_opencl_filter_frame(A @@ -386,72 +1212,50 @@ static int tonemap_opencl_filter_frame(A
ctx->range_out = output->color_range; ctx->range_out = output->color_range;
ctx->chroma_loc = output->chroma_location; ctx->chroma_loc = output->chroma_location;
@@ -3016,7 +3165,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
av_frame_free(&input); av_frame_free(&input);
av_frame_free(&output); av_frame_free(&output);
return err; return err;
@@ -458,62 +1141,101 @@ fail: @@ -459,62 +1263,101 @@ fail:
static av_cold void tonemap_opencl_uninit(AVFilterContext *avctx) static av_cold void tonemap_opencl_uninit(AVFilterContext *avctx)
{ {
@@ -3165,10 +3314,14 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
{ NULL } { NULL }
}; };
@@ -541,11 +1263,12 @@ const AVFilter ff_vf_tonemap_opencl = { @@ -540,13 +1383,14 @@ static const AVFilterPad tonemap_opencl_
.description = NULL_IF_CONFIG_SMALL("Perform HDR to SDR conversion with tonemapping."), const FFFilter ff_vf_tonemap_opencl = {
.p.name = "tonemap_opencl",
.p.description = NULL_IF_CONFIG_SMALL("Perform HDR to SDR conversion with tonemapping."),
- .p.priv_class = &tonemap_opencl_class,
- .p.flags = AVFILTER_FLAG_HWDEVICE,
.priv_size = sizeof(TonemapOpenCLContext), .priv_size = sizeof(TonemapOpenCLContext),
.priv_class = &tonemap_opencl_class, + .p.priv_class = &tonemap_opencl_class,
+ .preinit = &tonemap_opencl_preinit, + .preinit = &tonemap_opencl_preinit,
.init = &ff_opencl_filter_init, .init = &ff_opencl_filter_init,
.uninit = &tonemap_opencl_uninit, .uninit = &tonemap_opencl_uninit,
@@ -3177,5 +3330,5 @@ Index: FFmpeg/libavfilter/vf_tonemap_opencl.c
- FILTER_SINGLE_PIXFMT(AV_PIX_FMT_OPENCL), - FILTER_SINGLE_PIXFMT(AV_PIX_FMT_OPENCL),
+ FILTER_QUERY_FUNC(tonemap_opencl_query_formats), + FILTER_QUERY_FUNC(tonemap_opencl_query_formats),
.flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE, .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE,
.flags = AVFILTER_FLAG_HWDEVICE, + .p.flags = AVFILTER_FLAG_HWDEVICE,
}; };
+20 -3
View File
@@ -321,7 +321,8 @@ Index: FFmpeg/libavfilter/vf_overlay_opencl.c
- av_log(avctx, AV_LOG_DEBUG, "Using kernel %s.\n", kernel); - av_log(avctx, AV_LOG_DEBUG, "Using kernel %s.\n", kernel);
+ av_bprint_init(&header, 2048, AV_BPRINT_SIZE_UNLIMITED); + av_bprint_init(&header, 2048, AV_BPRINT_SIZE_UNLIMITED);
+
- err = ff_opencl_filter_load_program(avctx, &source, 1);
+ if (ctx->alpha && ctx->alpha_format == 1) + if (ctx->alpha && ctx->alpha_format == 1)
+ av_bprintf(&header, "#define NEED_UNPREMUL\n"); + av_bprintf(&header, "#define NEED_UNPREMUL\n");
+ +
@@ -329,8 +330,7 @@ Index: FFmpeg/libavfilter/vf_overlay_opencl.c
+ opencl_sources[0] = header.str; + opencl_sources[0] = header.str;
+ opencl_sources[1] = ff_source_overlay_cl; + opencl_sources[1] = ff_source_overlay_cl;
+ err = ff_opencl_filter_load_program(avctx, opencl_sources, OPENCL_SOURCE_NB); + err = ff_opencl_filter_load_program(avctx, opencl_sources, OPENCL_SOURCE_NB);
+
- err = ff_opencl_filter_load_program(avctx, &source, 1);
+ av_bprint_finalize(&header, NULL); + av_bprint_finalize(&header, NULL);
if (err < 0) if (err < 0)
goto fail; goto fail;
@@ -685,3 +685,20 @@ Index: FFmpeg/libavfilter/vf_overlay_opencl.c
{ NULL }, { NULL },
}; };
@@ -313,9 +514,8 @@ static const AVFilterPad overlay_opencl_
const FFFilter ff_vf_overlay_opencl = {
.p.name = "overlay_opencl",
.p.description = NULL_IF_CONFIG_SMALL("Overlay one video on top of another"),
- .p.priv_class = &overlay_opencl_class,
- .p.flags = AVFILTER_FLAG_HWDEVICE,
.priv_size = sizeof(OverlayOpenCLContext),
+ .p.priv_class = &overlay_opencl_class,
.init = &overlay_opencl_init,
.uninit = &overlay_opencl_uninit,
.activate = &overlay_opencl_activate,
@@ -323,4 +523,5 @@ const FFFilter ff_vf_overlay_opencl = {
FILTER_OUTPUTS(overlay_opencl_outputs),
FILTER_SINGLE_PIXFMT(AV_PIX_FMT_OPENCL),
.flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE,
+ .p.flags = AVFILTER_FLAG_HWDEVICE,
};
@@ -2,7 +2,7 @@ Index: FFmpeg/libavfilter/avfilter.h
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/avfilter.h --- FFmpeg.orig/libavfilter/avfilter.h
+++ FFmpeg/libavfilter/avfilter.h +++ FFmpeg/libavfilter/avfilter.h
@@ -579,6 +579,8 @@ struct AVFilterLink { @@ -417,6 +417,8 @@ struct AVFilterLink {
enum AVColorSpace colorspace; ///< agreed upon YUV color space enum AVColorSpace colorspace; ///< agreed upon YUV color space
enum AVColorRange color_range; ///< agreed upon YUV color range enum AVColorRange color_range; ///< agreed upon YUV color range
@@ -15,7 +15,7 @@ Index: FFmpeg/libavfilter/opencl.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/opencl.c --- FFmpeg.orig/libavfilter/opencl.c
+++ FFmpeg/libavfilter/opencl.c +++ FFmpeg/libavfilter/opencl.c
@@ -77,6 +77,9 @@ int ff_opencl_filter_config_input(AVFilt @@ -78,6 +78,9 @@ int ff_opencl_filter_config_input(AVFilt
if (!ctx->output_height) if (!ctx->output_height)
ctx->output_height = inlink->h; ctx->output_height = inlink->h;
@@ -25,7 +25,7 @@ Index: FFmpeg/libavfilter/opencl.c
return 0; return 0;
} }
@@ -126,6 +129,9 @@ int ff_opencl_filter_config_output(AVFil @@ -127,6 +130,9 @@ int ff_opencl_filter_config_output(AVFil
outlink->w = ctx->output_width; outlink->w = ctx->output_width;
outlink->h = ctx->output_height; outlink->h = ctx->output_height;
@@ -35,6 +35,29 @@ Index: FFmpeg/libavfilter/opencl.c
return 0; return 0;
fail: fail:
av_buffer_unref(&output_frames_ref); av_buffer_unref(&output_frames_ref);
Index: FFmpeg/libavfilter/vf_crop.c
===================================================================
--- FFmpeg.orig/libavfilter/vf_crop.c
+++ FFmpeg/libavfilter/vf_crop.c
@@ -152,6 +152,8 @@ static int config_input(AVFilterLink *li
if (pix_desc->flags & AV_PIX_FMT_FLAG_HWACCEL) {
s->hsub = 1;
s->vsub = 1;
+ if (ctx->nb_outputs > 0)
+ ctx->outputs[0]->fixed_pool_size = link->fixed_pool_size;
} else {
s->hsub = pix_desc->log2_chroma_w;
s->vsub = pix_desc->log2_chroma_h;
@@ -238,6 +240,9 @@ static int config_output(AVFilterLink *l
if (desc->flags & AV_PIX_FMT_FLAG_HWACCEL) {
// Hardware frames adjust the cropping regions rather than
// changing the frame size.
+ AVFilterContext *ctx = link->src;
+ if (ctx->nb_inputs > 0)
+ link->fixed_pool_size = ctx->inputs[0]->fixed_pool_size;
} else {
link->w = s->w;
link->h = s->h;
Index: FFmpeg/libavfilter/vf_hwmap.c Index: FFmpeg/libavfilter/vf_hwmap.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/vf_hwmap.c --- FFmpeg.orig/libavfilter/vf_hwmap.c
@@ -50,7 +73,7 @@ Index: FFmpeg/libavfilter/vf_hwmap.c
#include "avfilter.h" #include "avfilter.h"
#include "filters.h" #include "filters.h"
#include "formats.h" #include "formats.h"
@@ -124,6 +128,12 @@ static int hwmap_config_output(AVFilterL @@ -126,6 +130,12 @@ static int hwmap_config_output(AVFilterL
goto fail; goto fail;
} }
@@ -63,7 +86,7 @@ Index: FFmpeg/libavfilter/vf_hwmap.c
} else if (inlink->format == hwfc->format && } else if (inlink->format == hwfc->format &&
(desc->flags & AV_PIX_FMT_FLAG_HWACCEL) && (desc->flags & AV_PIX_FMT_FLAG_HWACCEL) &&
ctx->reverse) { ctx->reverse) {
@@ -133,6 +143,9 @@ static int hwmap_config_output(AVFilterL @@ -135,6 +145,9 @@ static int hwmap_config_output(AVFilterL
// mapped from that back to the source type. // mapped from that back to the source type.
AVBufferRef *source; AVBufferRef *source;
AVHWFramesContext *frames; AVHWFramesContext *frames;
@@ -73,7 +96,7 @@ Index: FFmpeg/libavfilter/vf_hwmap.c
ctx->hwframes_ref = av_hwframe_ctx_alloc(device); ctx->hwframes_ref = av_hwframe_ctx_alloc(device);
if (!ctx->hwframes_ref) { if (!ctx->hwframes_ref) {
@@ -146,8 +159,19 @@ static int hwmap_config_output(AVFilterL @@ -148,8 +161,19 @@ static int hwmap_config_output(AVFilterL
frames->width = hwfc->width; frames->width = hwfc->width;
frames->height = hwfc->height; frames->height = hwfc->height;
@@ -110,7 +133,7 @@ Index: FFmpeg/libavfilter/vf_hwupload.c
#include "avfilter.h" #include "avfilter.h"
#include "filters.h" #include "filters.h"
#include "formats.h" #include "formats.h"
@@ -112,6 +116,9 @@ static int hwupload_config_output(AVFilt @@ -119,6 +123,9 @@ static int hwupload_config_output(AVFilt
FilterLink *inl = ff_filter_link(inlink); FilterLink *inl = ff_filter_link(inlink);
HWUploadContext *ctx = avctx->priv; HWUploadContext *ctx = avctx->priv;
int err; int err;
@@ -120,7 +143,7 @@ Index: FFmpeg/libavfilter/vf_hwupload.c
av_buffer_unref(&ctx->hwframes_ref); av_buffer_unref(&ctx->hwframes_ref);
@@ -153,6 +160,11 @@ static int hwupload_config_output(AVFilt @@ -160,6 +167,11 @@ static int hwupload_config_output(AVFilt
if (avctx->extra_hw_frames >= 0) if (avctx->extra_hw_frames >= 0)
ctx->hwframes->initial_pool_size = 2 + avctx->extra_hw_frames; ctx->hwframes->initial_pool_size = 2 + avctx->extra_hw_frames;
@@ -136,7 +159,7 @@ Index: FFmpeg/libavutil/hwcontext_d3d11va.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavutil/hwcontext_d3d11va.c --- FFmpeg.orig/libavutil/hwcontext_d3d11va.c
+++ FFmpeg/libavutil/hwcontext_d3d11va.c +++ FFmpeg/libavutil/hwcontext_d3d11va.c
@@ -229,7 +229,7 @@ static AVBufferRef *d3d11va_alloc_single @@ -235,7 +235,7 @@ static AVBufferRef *d3d11va_alloc_single
.ArraySize = 1, .ArraySize = 1,
.Usage = D3D11_USAGE_DEFAULT, .Usage = D3D11_USAGE_DEFAULT,
.BindFlags = hwctx->BindFlags, .BindFlags = hwctx->BindFlags,
@@ -145,7 +168,7 @@ Index: FFmpeg/libavutil/hwcontext_d3d11va.c
}; };
hr = ID3D11Device_CreateTexture2D(device_hwctx->device, &texDesc, NULL, &tex); hr = ID3D11Device_CreateTexture2D(device_hwctx->device, &texDesc, NULL, &tex);
@@ -293,9 +293,17 @@ static int d3d11va_frames_init(AVHWFrame @@ -304,9 +304,17 @@ static int d3d11va_frames_init(AVHWFrame
.ArraySize = ctx->initial_pool_size, .ArraySize = ctx->initial_pool_size,
.Usage = D3D11_USAGE_DEFAULT, .Usage = D3D11_USAGE_DEFAULT,
.BindFlags = hwctx->BindFlags, .BindFlags = hwctx->BindFlags,
@@ -164,3 +187,12 @@ Index: FFmpeg/libavutil/hwcontext_d3d11va.c
if (hwctx->texture) { if (hwctx->texture) {
D3D11_TEXTURE2D_DESC texDesc2; D3D11_TEXTURE2D_DESC texDesc2;
ID3D11Texture2D_GetDesc(hwctx->texture, &texDesc2); ID3D11Texture2D_GetDesc(hwctx->texture, &texDesc2);
@@ -321,7 +329,7 @@ static int d3d11va_frames_init(AVHWFrame
ctx->initial_pool_size = texDesc2.ArraySize;
hwctx->BindFlags = texDesc2.BindFlags;
hwctx->MiscFlags = texDesc2.MiscFlags;
- } else if (texDesc.ArraySize > 0) {
+ } else if (!(texDesc.BindFlags & D3D11_BIND_RENDER_TARGET) && texDesc.ArraySize > 0) {
hr = ID3D11Device_CreateTexture2D(device_hwctx->device, &texDesc, NULL, &hwctx->texture);
if (FAILED(hr)) {
av_log(ctx, AV_LOG_ERROR, "Could not create the texture (%lx)\n", (long)hr);
@@ -2,8 +2,8 @@ Index: FFmpeg/libavutil/hwcontext.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavutil/hwcontext.c --- FFmpeg.orig/libavutil/hwcontext.c
+++ FFmpeg/libavutil/hwcontext.c +++ FFmpeg/libavutil/hwcontext.c
@@ -84,21 +84,6 @@ static const char *const hw_type_names[] @@ -92,21 +92,6 @@ static const char *const hw_type_names[]
[AV_HWDEVICE_TYPE_VULKAN] = "vulkan", [AV_HWDEVICE_TYPE_OHCODEC] = "ohcodec",
}; };
-typedef struct FFHWDeviceContext { -typedef struct FFHWDeviceContext {
@@ -24,7 +24,7 @@ Index: FFmpeg/libavutil/hwcontext.c
enum AVHWDeviceType av_hwdevice_find_type_by_name(const char *name) enum AVHWDeviceType av_hwdevice_find_type_by_name(const char *name)
{ {
int type; int type;
@@ -143,6 +128,7 @@ static void hwdevice_ctx_free(void *opaq @@ -158,6 +143,7 @@ static void hwdevice_ctx_free(void *opaq
{ {
FFHWDeviceContext *ctxi = (FFHWDeviceContext*)data; FFHWDeviceContext *ctxi = (FFHWDeviceContext*)data;
AVHWDeviceContext *ctx = &ctxi->p; AVHWDeviceContext *ctx = &ctxi->p;
@@ -32,7 +32,7 @@ Index: FFmpeg/libavutil/hwcontext.c
/* uninit might still want access the hw context and the user /* uninit might still want access the hw context and the user
* free() callback might destroy it, so uninit has to be called first */ * free() callback might destroy it, so uninit has to be called first */
@@ -153,6 +139,8 @@ static void hwdevice_ctx_free(void *opaq @@ -168,6 +154,8 @@ static void hwdevice_ctx_free(void *opaq
ctx->free(ctx); ctx->free(ctx);
av_buffer_unref(&ctxi->source_device); av_buffer_unref(&ctxi->source_device);
@@ -41,7 +41,7 @@ Index: FFmpeg/libavutil/hwcontext.c
av_freep(&ctx->hwctx); av_freep(&ctx->hwctx);
av_freep(&ctx); av_freep(&ctx);
@@ -633,6 +621,28 @@ fail: @@ -648,6 +636,28 @@ fail:
return ret; return ret;
} }
@@ -70,7 +70,7 @@ Index: FFmpeg/libavutil/hwcontext.c
int av_hwdevice_ctx_create_derived_opts(AVBufferRef **dst_ref_ptr, int av_hwdevice_ctx_create_derived_opts(AVBufferRef **dst_ref_ptr,
enum AVHWDeviceType type, enum AVHWDeviceType type,
AVBufferRef *src_ref, AVBufferRef *src_ref,
@@ -656,6 +666,16 @@ int av_hwdevice_ctx_create_derived_opts( @@ -671,6 +681,16 @@ int av_hwdevice_ctx_create_derived_opts(
tmp_ref = tmp_ctx->source_device; tmp_ref = tmp_ctx->source_device;
} }
@@ -87,7 +87,7 @@ Index: FFmpeg/libavutil/hwcontext.c
dst_ref = av_hwdevice_ctx_alloc(type); dst_ref = av_hwdevice_ctx_alloc(type);
if (!dst_ref) { if (!dst_ref) {
ret = AVERROR(ENOMEM); ret = AVERROR(ENOMEM);
@@ -676,6 +696,11 @@ int av_hwdevice_ctx_create_derived_opts( @@ -691,6 +711,11 @@ int av_hwdevice_ctx_create_derived_opts(
ret = AVERROR(ENOMEM); ret = AVERROR(ENOMEM);
goto fail; goto fail;
} }
@@ -103,10 +103,10 @@ Index: FFmpeg/libavutil/hwcontext.h
=================================================================== ===================================================================
--- FFmpeg.orig/libavutil/hwcontext.h --- FFmpeg.orig/libavutil/hwcontext.h
+++ FFmpeg/libavutil/hwcontext.h +++ FFmpeg/libavutil/hwcontext.h
@@ -38,6 +38,7 @@ enum AVHWDeviceType { @@ -41,6 +41,7 @@ enum AVHWDeviceType {
AV_HWDEVICE_TYPE_MEDIACODEC, AV_HWDEVICE_TYPE_AMF,
AV_HWDEVICE_TYPE_VULKAN, /* OpenHarmony Codec device */
AV_HWDEVICE_TYPE_D3D12VA, AV_HWDEVICE_TYPE_OHCODEC,
+ AV_HWDEVICE_TYPE_NB, ///< number of hw device types, not part of API/ABI. + AV_HWDEVICE_TYPE_NB, ///< number of hw device types, not part of API/ABI.
}; };
@@ -115,9 +115,9 @@ Index: FFmpeg/libavutil/hwcontext_internal.h
=================================================================== ===================================================================
--- FFmpeg.orig/libavutil/hwcontext_internal.h --- FFmpeg.orig/libavutil/hwcontext_internal.h
+++ FFmpeg/libavutil/hwcontext_internal.h +++ FFmpeg/libavutil/hwcontext_internal.h
@@ -164,4 +164,25 @@ extern const HWContextType ff_hwcontext_ @@ -166,4 +166,25 @@ extern const HWContextType ff_hwcontext_
extern const HWContextType ff_hwcontext_type_mediacodec; extern const HWContextType ff_hwcontext_type_amf;
extern const HWContextType ff_hwcontext_type_vulkan; extern const HWContextType ff_hwcontext_type_oh;
+typedef struct FFHWDeviceContext { +typedef struct FFHWDeviceContext {
+ /** + /**
+27 -25
View File
@@ -2,8 +2,8 @@ Index: FFmpeg/configure
=================================================================== ===================================================================
--- FFmpeg.orig/configure --- FFmpeg.orig/configure
+++ FFmpeg/configure +++ FFmpeg/configure
@@ -3884,6 +3884,7 @@ gblur_vulkan_filter_deps="vulkan spirv_c @@ -4143,6 +4143,7 @@ gblur_vulkan_filter_deps="vulkan spirv_l
hflip_vulkan_filter_deps="vulkan spirv_compiler" hflip_vulkan_filter_deps="vulkan spirv_library"
histeq_filter_deps="gpl" histeq_filter_deps="gpl"
hqdn3d_filter_deps="gpl" hqdn3d_filter_deps="gpl"
+hwupload_vaapi_filter_deps="vaapi" +hwupload_vaapi_filter_deps="vaapi"
@@ -14,7 +14,7 @@ Index: FFmpeg/libavfilter/Makefile
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/Makefile --- FFmpeg.orig/libavfilter/Makefile
+++ FFmpeg/libavfilter/Makefile +++ FFmpeg/libavfilter/Makefile
@@ -347,6 +347,7 @@ OBJS-$(CONFIG_HUESATURATION_FILTER) @@ -355,6 +355,7 @@ OBJS-$(CONFIG_HUESATURATION_FILTER)
OBJS-$(CONFIG_HWDOWNLOAD_FILTER) += vf_hwdownload.o OBJS-$(CONFIG_HWDOWNLOAD_FILTER) += vf_hwdownload.o
OBJS-$(CONFIG_HWMAP_FILTER) += vf_hwmap.o OBJS-$(CONFIG_HWMAP_FILTER) += vf_hwmap.o
OBJS-$(CONFIG_HWUPLOAD_CUDA_FILTER) += vf_hwupload_cuda.o OBJS-$(CONFIG_HWUPLOAD_CUDA_FILTER) += vf_hwupload_cuda.o
@@ -26,19 +26,19 @@ Index: FFmpeg/libavfilter/allfilters.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/allfilters.c --- FFmpeg.orig/libavfilter/allfilters.c
+++ FFmpeg/libavfilter/allfilters.c +++ FFmpeg/libavfilter/allfilters.c
@@ -324,6 +324,7 @@ extern const AVFilter ff_vf_hwdownload; @@ -330,6 +330,7 @@ extern const FFFilter ff_vf_hwdownload;
extern const AVFilter ff_vf_hwmap; extern const FFFilter ff_vf_hwmap;
extern const AVFilter ff_vf_hwupload; extern const FFFilter ff_vf_hwupload;
extern const AVFilter ff_vf_hwupload_cuda; extern const FFFilter ff_vf_hwupload_cuda;
+extern const AVFilter ff_vf_hwupload_vaapi; +extern const FFFilter ff_vf_hwupload_vaapi;
extern const AVFilter ff_vf_hysteresis; extern const FFFilter ff_vf_hysteresis;
extern const AVFilter ff_vf_iccdetect; extern const FFFilter ff_vf_iccdetect;
extern const AVFilter ff_vf_iccgen; extern const FFFilter ff_vf_iccgen;
Index: FFmpeg/libavfilter/vf_hwupload_vaapi.c Index: FFmpeg/libavfilter/vf_hwupload_vaapi.c
=================================================================== ===================================================================
--- /dev/null --- /dev/null
+++ FFmpeg/libavfilter/vf_hwupload_vaapi.c +++ FFmpeg/libavfilter/vf_hwupload_vaapi.c
@@ -0,0 +1,195 @@ @@ -0,0 +1,197 @@
+/* +/*
+ * This file is part of FFmpeg. + * This file is part of FFmpeg.
+ * + *
@@ -89,7 +89,9 @@ Index: FFmpeg/libavfilter/vf_hwupload_vaapi.c
+ av_buffer_unref(&s->hwdevice); + av_buffer_unref(&s->hwdevice);
+} +}
+ +
+static int vaapiupload_query_formats(AVFilterContext *ctx) +static int vaapiupload_query_formats(const AVFilterContext *ctx,
+ AVFilterFormatsConfig **cfg_in,
+ AVFilterFormatsConfig **cfg_out)
+{ +{
+ int ret; + int ret;
+ +
@@ -105,16 +107,16 @@ Index: FFmpeg/libavfilter/vf_hwupload_vaapi.c
+ static const enum AVPixelFormat output_pix_fmts[] = { + static const enum AVPixelFormat output_pix_fmts[] = {
+ AV_PIX_FMT_VAAPI, AV_PIX_FMT_NONE, + AV_PIX_FMT_VAAPI, AV_PIX_FMT_NONE,
+ }; + };
+ AVFilterFormats *in_fmts = ff_make_format_list(input_pix_fmts); + AVFilterFormats *in_fmts = ff_make_pixel_format_list(input_pix_fmts);
+ AVFilterFormats *out_fmts; + AVFilterFormats *out_fmts;
+ +
+ ret = ff_formats_ref(in_fmts, &ctx->inputs[0]->outcfg.formats); + ret = ff_formats_ref(in_fmts, &cfg_in[0]->formats);
+ if (ret < 0) + if (ret < 0)
+ return ret; + return ret;
+ +
+ out_fmts = ff_make_format_list(output_pix_fmts); + out_fmts = ff_make_pixel_format_list(output_pix_fmts);
+ +
+ ret = ff_formats_ref(out_fmts, &ctx->outputs[0]->incfg.formats); + ret = ff_formats_ref(out_fmts, &cfg_out[0]->formats);
+ if (ret < 0) + if (ret < 0)
+ return ret; + return ret;
+ +
@@ -218,19 +220,19 @@ Index: FFmpeg/libavfilter/vf_hwupload_vaapi.c
+ }, + },
+}; +};
+ +
+const AVFilter ff_vf_hwupload_vaapi = { +const FFFilter ff_vf_hwupload_vaapi = {
+ .name = "hwupload_vaapi", + .p.name = "hwupload_vaapi",
+ .description = NULL_IF_CONFIG_SMALL("Upload a system memory frame to a VAAPI device."), + .p.description = NULL_IF_CONFIG_SMALL("Upload a system memory frame to a VAAPI device."),
+ +
+ .init = vaapiupload_init, + .init = vaapiupload_init,
+ .uninit = vaapiupload_uninit, + .uninit = vaapiupload_uninit,
+ +
+ .priv_size = sizeof(VaapiUploadContext), + .priv_size = sizeof(VaapiUploadContext),
+ .priv_class = &vaapiupload_class, + .p.priv_class = &vaapiupload_class,
+ +
+ FILTER_INPUTS(vaapiupload_inputs), + FILTER_INPUTS(vaapiupload_inputs),
+ FILTER_OUTPUTS(vaapiupload_outputs), + FILTER_OUTPUTS(vaapiupload_outputs),
+ FILTER_QUERY_FUNC(vaapiupload_query_formats), + FILTER_QUERY_FUNC2(vaapiupload_query_formats),
+ +
+ .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE, + .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE,
+}; +};
@@ -1,43 +0,0 @@
Index: FFmpeg/libavfilter/vf_overlay_vaapi.c
===================================================================
--- FFmpeg.orig/libavfilter/vf_overlay_vaapi.c
+++ FFmpeg/libavfilter/vf_overlay_vaapi.c
@@ -53,6 +53,7 @@ typedef struct OverlayVAAPIContext {
int ow;
int oh;
float alpha;
+ int alpha_format;
unsigned int blend_flags;
float blend_alpha;
} OverlayVAAPIContext;
@@ -289,6 +290,7 @@ static int overlay_vaapi_config_input_ma
static int overlay_vaapi_config_input_overlay(AVFilterLink *inlink)
{
AVFilterContext *avctx = inlink->dst;
+ AVFilterLink *inlink0 = avctx->inputs[0];
OverlayVAAPIContext *ctx = avctx->priv;
int ret;
@@ -312,7 +314,11 @@ static int overlay_vaapi_config_input_ov
ctx->blend_alpha = ctx->alpha;
}
- if (have_alpha_planar(inlink))
+ // VA_BLEND_PREMULTIPLIED_ALPHA requires both inputs
+ // have alpha planar, otherwise it just fails.
+ if (have_alpha_planar(inlink) &&
+ have_alpha_planar(inlink0) &&
+ ctx->alpha_format == 1)
ctx->blend_flags |= VA_BLEND_PREMULTIPLIED_ALPHA;
return 0;
@@ -380,6 +386,9 @@ static const AVOption overlay_vaapi_opti
{ "w", "Overlay width", OFFSET(overlay_ow), AV_OPT_TYPE_STRING, { .str="overlay_iw"}, 0, 255, .flags = FLAGS},
{ "h", "Overlay height", OFFSET(overlay_oh), AV_OPT_TYPE_STRING, { .str="overlay_ih*w/overlay_iw"}, 0, 255, .flags = FLAGS},
{ "alpha", "Overlay global alpha", OFFSET(alpha), AV_OPT_TYPE_FLOAT, { .dbl = 1.0 }, 0.0, 1.0, .flags = FLAGS },
+ { "alpha_format", "alpha format", OFFSET(alpha_format), AV_OPT_TYPE_INT, { .i64 = 0 }, 0, 1, FLAGS, .unit = "alpha_format" },
+ { "straight", "The overlay input is unpremultiplied", 0, AV_OPT_TYPE_CONST, { .i64 = 0 }, .flags = FLAGS, .unit = "alpha_format" },
+ { "premultiplied", "The overlay input is premultiplied", 0, AV_OPT_TYPE_CONST, { .i64 = 1 }, .flags = FLAGS, .unit = "alpha_format" },
{ "eof_action", "Action to take when encountering EOF from secondary input ",
OFFSET(fs.opt_eof_action), AV_OPT_TYPE_INT, { .i64 = EOF_ACTION_REPEAT },
EOF_ACTION_REPEAT, EOF_ACTION_PASS, .flags = FLAGS, .unit = "eof_action" },
@@ -0,0 +1,46 @@
Index: FFmpeg/libavfilter/vf_overlay_vaapi.c
===================================================================
--- FFmpeg.orig/libavfilter/vf_overlay_vaapi.c
+++ FFmpeg/libavfilter/vf_overlay_vaapi.c
@@ -194,6 +194,14 @@ static int overlay_vaapi_blend(FFFrameSy
if (err < 0)
goto fail;
+ if (input_overlay) {
+ /* call vaapi_vpp_colour_properties() for overlay */
+ err = ff_vaapi_vpp_init_params(avctx, &params[1],
+ input_overlay, output);
+ if (err < 0)
+ goto fail;
+ }
+
err = ff_vaapi_vpp_init_params(avctx, &params[0],
input_main, output);
if (err < 0)
@@ -228,8 +236,6 @@ static int overlay_vaapi_blend(FFFrameSy
"will crop the overlay image according based on the main image.\n");
}
- memcpy(&params[1], &params[0], sizeof(params[0]));
-
blend_state.flags = ctx->blend_flags;
blend_state.global_alpha = ctx->blend_alpha;
params[1].blend_state = &blend_state;
@@ -237,6 +243,7 @@ static int overlay_vaapi_blend(FFFrameSy
params[1].surface = (VASurfaceID)(uintptr_t)input_overlay->data[3];
params[1].surface_region = NULL;
params[1].output_region = &overlay_region;
+ params[1].output_background_color = 0;
}
err = ff_vaapi_vpp_render_pictures(avctx, params, input_overlay ? 2 : 1, output);
@@ -312,7 +319,8 @@ static int overlay_vaapi_config_input_ov
ctx->blend_alpha = ctx->alpha;
}
- if (have_alpha_planar(inlink))
+ if (have_alpha_planar(inlink) &&
+ inlink->alpha_mode == AVALPHA_MODE_PREMULTIPLIED)
ctx->blend_flags |= VA_BLEND_PREMULTIPLIED_ALPHA;
return 0;
@@ -2,7 +2,7 @@ Index: FFmpeg/libavcodec/nvdec.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavcodec/nvdec.c --- FFmpeg.orig/libavcodec/nvdec.c
+++ FFmpeg/libavcodec/nvdec.c +++ FFmpeg/libavcodec/nvdec.c
@@ -399,8 +399,8 @@ int ff_nvdec_decode_init(AVCodecContext @@ -413,8 +413,8 @@ int ff_nvdec_decode_init(AVCodecContext
params.OutputFormat = output_format; params.OutputFormat = output_format;
params.CodecType = cuvid_codec_type; params.CodecType = cuvid_codec_type;
params.ChromaFormat = cuvid_chroma_format; params.ChromaFormat = cuvid_chroma_format;
@@ -13,20 +13,20 @@ Index: FFmpeg/libavcodec/nvdec.c
ret = nvdec_decoder_create(&ctx->decoder, frames_ctx->device_ref, &params, avctx); ret = nvdec_decoder_create(&ctx->decoder, frames_ctx->device_ref, &params, avctx);
if (ret < 0) { if (ret < 0) {
@@ -424,7 +424,7 @@ int ff_nvdec_decode_init(AVCodecContext @@ -438,7 +438,7 @@ int ff_nvdec_decode_init(AVCodecContext
ret = AVERROR(ENOMEM); ret = AVERROR(ENOMEM);
goto fail; goto fail;
} }
- pool->dpb_size = frames_ctx->initial_pool_size; - pool->dpb_size = frames_ctx->initial_pool_size;
+ pool->dpb_size = FFMIN(frames_ctx->initial_pool_size, 32); + pool->dpb_size = FFMIN(frames_ctx->initial_pool_size, 32);
ctx->decoder_pool = ff_refstruct_pool_alloc_ext(sizeof(unsigned int), 0, pool, ctx->decoder_pool = av_refstruct_pool_alloc_ext(sizeof(unsigned int), 0, pool,
nvdec_decoder_frame_init, nvdec_decoder_frame_init,
@@ -529,7 +529,6 @@ static int nvdec_retrieve_data(void *log @@ -543,7 +543,6 @@ static int nvdec_retrieve_data(void *log
goto copy_fail; goto copy_fail;
unmap_data->idx = cf->idx; unmap_data->idx = cf->idx;
- unmap_data->idx_ref = ff_refstruct_ref(cf->idx_ref); - unmap_data->idx_ref = av_refstruct_ref(cf->idx_ref);
unmap_data->decoder = ff_refstruct_ref(cf->decoder); unmap_data->decoder = av_refstruct_ref(cf->decoder);
av_pix_fmt_get_chroma_sub_sample(hwctx->sw_format, &shift_h, &shift_v); av_pix_fmt_get_chroma_sub_sample(hwctx->sw_format, &shift_h, &shift_v);
@@ -0,0 +1,54 @@
Index: FFmpeg/libavcodec/qsvenc.c
===================================================================
--- FFmpeg.orig/libavcodec/qsvenc.c
+++ FFmpeg/libavcodec/qsvenc.c
@@ -1840,6 +1840,9 @@ int ff_qsv_enc_init(AVCodecContext *avct
return ret;
}
+ // Update AVCodecContext with actual encoding parameters
+ avctx->has_b_frames = q->param.mfx.GopRefDist > 1 ? q->param.mfx.GopRefDist - 1 : 0;
+
q->avctx = avctx;
return 0;
@@ -2497,7 +2500,8 @@ static int encode_frame(AVCodecContext *
pkt.bs->Data = pkt.pkt.data;
pkt.bs->MaxLength = pkt.pkt.size;
- if (avctx->codec_id == AV_CODEC_ID_H264) {
+ if (avctx->codec_id == AV_CODEC_ID_H264 ||
+ avctx->codec_id == AV_CODEC_ID_HEVC) {
enc_info = av_mallocz(sizeof(*enc_info));
if (!enc_info)
goto nomem;
@@ -2563,7 +2567,8 @@ free:
av_freep(&pkt.sync);
av_packet_unref(&pkt.pkt);
av_freep(&pkt.bs);
- if (avctx->codec_id == AV_CODEC_ID_H264) {
+ if (avctx->codec_id == AV_CODEC_ID_H264 ||
+ avctx->codec_id == AV_CODEC_ID_HEVC) {
av_freep(&enc_info);
av_freep(&enc_buf);
}
@@ -2684,7 +2689,8 @@ int ff_qsv_encode(AVCodecContext *avctx,
return AVERROR_INVALIDDATA;
}
- if (avctx->codec_id == AV_CODEC_ID_H264) {
+ if (avctx->codec_id == AV_CODEC_ID_H264 ||
+ avctx->codec_id == AV_CODEC_ID_HEVC) {
enc_buf = qpkt.bs->ExtParam;
enc_info = (mfxExtAVCEncodedFrameInfo *)(*enc_buf);
ff_encode_add_stats_side_data(&qpkt.pkt,
@@ -2728,7 +2734,8 @@ int ff_qsv_enc_close(AVCodecContext *avc
if (q->async_fifo) {
QSVPacket pkt;
while (av_fifo_read(q->async_fifo, &pkt, 1) >= 0) {
- if (avctx->codec_id == AV_CODEC_ID_H264) {
+ if (avctx->codec_id == AV_CODEC_ID_H264 ||
+ avctx->codec_id == AV_CODEC_ID_HEVC) {
mfxExtBuffer **enc_buf = pkt.bs->ExtParam;
mfxExtAVCEncodedFrameInfo *enc_info = (mfxExtAVCEncodedFrameInfo *)(*enc_buf);
av_freep(&enc_info);
@@ -1,144 +0,0 @@
Index: FFmpeg/libavutil/hwcontext_vaapi.c
===================================================================
--- FFmpeg.orig/libavutil/hwcontext_vaapi.c
+++ FFmpeg/libavutil/hwcontext_vaapi.c
@@ -1813,6 +1813,7 @@ static int vaapi_device_create(AVHWDevic
"Failed to get DRM device info for device %d.\n", n);
close(priv->drm_fd);
priv->drm_fd = -1;
+ drmFreeVersion(info);
continue;
}
@@ -1824,6 +1825,7 @@ static int vaapi_device_create(AVHWDevic
drmFreeDevice(&device);
close(priv->drm_fd);
priv->drm_fd = -1;
+ drmFreeVersion(info);
continue;
}
av_log(ctx, AV_LOG_VERBOSE, "Trying to use "
@@ -1831,6 +1833,7 @@ static int vaapi_device_create(AVHWDevic
"with matching vendor id (%s).\n",
n, vendor_id->value);
drmFreeDevice(&device);
+ drmFreeVersion(info);
break;
}
drmFreeVersion(info);
Index: FFmpeg/libavcodec/vaapi_encode.c
===================================================================
--- FFmpeg.orig/libavcodec/vaapi_encode.c
+++ FFmpeg/libavcodec/vaapi_encode.c
@@ -1133,6 +1133,68 @@ fail:
return err;
}
+static av_cold int vaapi_encode_surface_alignment(av_unused AVCodecContext *avctx)
+{
+#if VA_CHECK_VERSION(1, 21, 0)
+ VAAPIEncodeContext *ctx = avctx->priv_data;
+ VASurfaceAttrib *attr_list = NULL;
+ unsigned int attr_count = 0;
+ VAConfigID va_config;
+ VAStatus vas;
+ int err = 0;
+
+ vas = vaCreateConfig(ctx->hwctx->display,
+ ctx->va_profile, ctx->va_entrypoint,
+ NULL, 0, &va_config);
+ if (vas != VA_STATUS_SUCCESS) {
+ av_log(avctx, AV_LOG_ERROR, "Failed to create temp encode pipeline "
+ "configuration: %d (%s).\n", vas, vaErrorStr(vas));
+ return AVERROR(EIO);
+ }
+
+ vas = vaQuerySurfaceAttributes(ctx->hwctx->display, va_config,
+ 0, &attr_count);
+ if (vas != VA_STATUS_SUCCESS) {
+ av_log(avctx, AV_LOG_ERROR, "Failed to query surface attributes: "
+ "%d (%s).\n", vas, vaErrorStr(vas));
+ err = AVERROR_EXTERNAL;
+ goto fail;
+ }
+
+ attr_list = av_malloc(attr_count * sizeof(*attr_list));
+ if (!attr_list) {
+ err = AVERROR(ENOMEM);
+ goto fail;
+ }
+
+ vas = vaQuerySurfaceAttributes(ctx->hwctx->display, va_config,
+ attr_list, &attr_count);
+ if (vas != VA_STATUS_SUCCESS) {
+ av_log(avctx, AV_LOG_ERROR, "Failed to query surface attributes: "
+ "%d (%s).\n", vas, vaErrorStr(vas));
+ err = AVERROR_EXTERNAL;
+ goto fail;
+ }
+
+ for (unsigned int i = 0; i < attr_count; i++) {
+ if (attr_list[i].type == VASurfaceAttribAlignmentSize) {
+ ctx->surface_alignment_width =
+ 1 << (attr_list[i].value.value.i & 0xf);
+ ctx->surface_alignment_height =
+ 1 << ((attr_list[i].value.value.i & 0xf0) >> 4);
+ break;
+ }
+ }
+
+fail:
+ av_freep(&attr_list);
+ vaDestroyConfig(ctx->hwctx->display, va_config);
+ return err;
+#else
+ return 0;
+#endif
+}
+
static const VAAPIEncodeRCMode vaapi_encode_rc_modes[] = {
// Bitrate Quality
// | Maxrate | HRD/VBV
@@ -2111,6 +2173,10 @@ av_cold int ff_vaapi_encode_init(AVCodec
if (err < 0)
goto fail;
+ err = vaapi_encode_surface_alignment(avctx);
+ if (err < 0)
+ goto fail;
+
if (ctx->codec->get_encoder_caps) {
err = ctx->codec->get_encoder_caps(avctx);
if (err < 0)
Index: FFmpeg/libavcodec/vaapi_encode.h
===================================================================
--- FFmpeg.orig/libavcodec/vaapi_encode.h
+++ FFmpeg/libavcodec/vaapi_encode.h
@@ -260,6 +260,10 @@ typedef struct VAAPIEncodeContext {
* This is a RefStruct reference.
*/
VABufferID *coded_buffer_ref;
+
+ // Surface alignment required by driver.
+ int surface_alignment_width;
+ int surface_alignment_height;
} VAAPIEncodeContext;
typedef struct VAAPIEncodeType {
Index: FFmpeg/libavcodec/vaapi_encode_h265.c
===================================================================
--- FFmpeg.orig/libavcodec/vaapi_encode_h265.c
+++ FFmpeg/libavcodec/vaapi_encode_h265.c
@@ -951,8 +951,10 @@ static av_cold int vaapi_encode_h265_get
"min CB size %dx%d.\n", priv->ctu_size, priv->ctu_size,
priv->min_cb_size, priv->min_cb_size);
- base_ctx->surface_width = FFALIGN(avctx->width, priv->min_cb_size);
- base_ctx->surface_height = FFALIGN(avctx->height, priv->min_cb_size);
+ base_ctx->surface_width = FFALIGN(avctx->width,
+ FFMAX(priv->min_cb_size, priv->common.surface_alignment_width));
+ base_ctx->surface_height = FFALIGN(avctx->height,
+ FFMAX(priv->min_cb_size, priv->common.surface_alignment_height));
base_ctx->slice_block_width = base_ctx->slice_block_height = priv->ctu_size;
@@ -30,7 +30,7 @@ Index: FFmpeg/libavcodec/qsvdec.c
#include "qsv.h" #include "qsv.h"
+#include "h264_sei.h" +#include "h264_sei.h"
#include "qsv_internal.h" #include "qsv_internal.h"
#include "refstruct.h" #include "libavutil/refstruct.h"
@@ -111,8 +114,13 @@ typedef struct QSVContext { @@ -111,8 +114,13 @@ typedef struct QSVContext {
@@ -207,7 +207,7 @@ Index: FFmpeg/libavcodec/qsvdec.c
if ((avctx->export_side_data & AV_CODEC_EXPORT_DATA_FILM_GRAIN) && if ((avctx->export_side_data & AV_CODEC_EXPORT_DATA_FILM_GRAIN) &&
@@ -986,6 +1139,8 @@ static void qsv_decode_close_qsvcontext( @@ -986,6 +1139,8 @@ static void qsv_decode_close_qsvcontext(
av_buffer_unref(&q->frames_ctx.hw_frames_ctx); av_buffer_unref(&q->frames_ctx.hw_frames_ctx);
ff_refstruct_unref(&q->frames_ctx.mids); av_refstruct_unref(&q->frames_ctx.mids);
av_buffer_pool_uninit(&q->pool); av_buffer_pool_uninit(&q->pool);
+ +
+ av_freep(&q->payload.Data); + av_freep(&q->payload.Data);
@@ -1,81 +0,0 @@
Index: FFmpeg/libavcodec/qsvdec.c
===================================================================
--- FFmpeg.orig/libavcodec/qsvdec.c
+++ FFmpeg/libavcodec/qsvdec.c
@@ -202,6 +202,7 @@ static int qsv_init_session(AVCodecConte
MFXClose(q->internal_qs.session);
q->internal_qs.session = NULL;
}
+ ff_refstruct_unref(&q->frames_ctx.mids);
av_buffer_unref(&q->frames_ctx.hw_frames_ctx);
q->frames_ctx.hw_frames_ctx = av_buffer_ref(hw_frames_ref);
Index: FFmpeg/libavcodec/qsvenc.c
===================================================================
--- FFmpeg.orig/libavcodec/qsvenc.c
+++ FFmpeg/libavcodec/qsvenc.c
@@ -1842,6 +1842,9 @@ int ff_qsv_enc_init(AVCodecContext *avct
return ret;
}
+ // Update AVCodecContext with actual encoding parameters
+ avctx->has_b_frames = q->param.mfx.GopRefDist > 1 ? q->param.mfx.GopRefDist - 1 : 0;
+
q->avctx = avctx;
return 0;
Index: FFmpeg/libavfilter/qsvvpp.c
===================================================================
--- FFmpeg.orig/libavfilter/qsvvpp.c
+++ FFmpeg/libavfilter/qsvvpp.c
@@ -471,7 +471,8 @@ static QSVFrame *submit_frame(QSVVPPCont
}
/* get the output surface */
-static QSVFrame *query_frame(QSVVPPContext *s, AVFilterLink *outlink, const AVFrame *in)
+static QSVFrame *query_frame(QSVVPPContext *s, AVFilterLink *outlink, const AVFrame *in,
+ const AVFrame *propref)
{
FilterLink *l = ff_filter_link(outlink);
AVFilterContext *ctx = outlink->src;
@@ -513,6 +514,15 @@ static QSVFrame *query_frame(QSVVPPConte
return NULL;
}
+ if (propref) {
+ ret = av_frame_copy_props(out_frame->frame, propref);
+ if (ret < 0) {
+ av_frame_free(&out_frame->frame);
+ av_log(ctx, AV_LOG_ERROR, "Failed to copy metadata fields from src to dst.\n");
+ return NULL;
+ }
+ }
+
if (l->frame_rate.num && l->frame_rate.den)
out_frame->frame->duration = av_rescale_q(1, av_inv_q(l->frame_rate), outlink->time_base);
else
@@ -985,7 +995,7 @@ int ff_qsvvpp_filter_frame(QSVVPPContext
}
do {
- out_frame = query_frame(s, outlink, in_frame->frame);
+ out_frame = query_frame(s, outlink, in_frame->frame, propref);
if (!out_frame) {
av_log(ctx, AV_LOG_ERROR, "Failed to query an output frame.\n");
return AVERROR(ENOMEM);
@@ -1009,15 +1019,6 @@ int ff_qsvvpp_filter_frame(QSVVPPContext
break;
}
- if (propref) {
- ret1 = av_frame_copy_props(out_frame->frame, propref);
- if (ret1 < 0) {
- av_frame_free(&out_frame->frame);
- av_log(ctx, AV_LOG_ERROR, "Failed to copy metadata fields from src to dst.\n");
- return ret1;
- }
- }
-
out_frame->frame->pts = av_rescale_q(out_frame->surface.Data.TimeStamp,
default_tb, outlink->time_base);
@@ -33,7 +33,7 @@ Index: FFmpeg/libavfilter/qsvvpp.c
else if (qsv_frame->frame->repeat_pict == 2) else if (qsv_frame->frame->repeat_pict == 2)
qsv_frame->surface.Info.PicStruct |= MFX_PICSTRUCT_FRAME_DOUBLING; qsv_frame->surface.Info.PicStruct |= MFX_PICSTRUCT_FRAME_DOUBLING;
else if (qsv_frame->frame->repeat_pict == 4) else if (qsv_frame->frame->repeat_pict == 4)
@@ -913,8 +918,13 @@ static int qsvvpp_init_vpp_session(AVFil @@ -908,8 +913,13 @@ static int qsvvpp_init_vpp_session(AVFil
/* Query VPP params again, including params for frame */ /* Query VPP params again, including params for frame */
ret = MFXVideoVPP_Query(s->session, &s->vpp_param, &s->vpp_param); ret = MFXVideoVPP_Query(s->session, &s->vpp_param, &s->vpp_param);
@@ -150,7 +150,7 @@ Index: FFmpeg/libavfilter/vf_overlay_qsv.c
vpp->qsv_param.out_sw_format = hw_frame0->sw_format; vpp->qsv_param.out_sw_format = hw_frame0->sw_format;
} }
@@ -374,6 +371,7 @@ static int overlay_qsv_query_formats(AVF @@ -376,6 +373,7 @@ static int overlay_qsv_query_formats(con
static const enum AVPixelFormat main_in_fmts[] = { static const enum AVPixelFormat main_in_fmts[] = {
AV_PIX_FMT_YUV420P, AV_PIX_FMT_YUV420P,
AV_PIX_FMT_NV12, AV_PIX_FMT_NV12,
@@ -158,7 +158,7 @@ Index: FFmpeg/libavfilter/vf_overlay_qsv.c
AV_PIX_FMT_YUYV422, AV_PIX_FMT_YUYV422,
AV_PIX_FMT_RGB32, AV_PIX_FMT_RGB32,
AV_PIX_FMT_QSV, AV_PIX_FMT_QSV,
@@ -381,6 +379,7 @@ static int overlay_qsv_query_formats(AVF @@ -383,6 +381,7 @@ static int overlay_qsv_query_formats(con
}; };
static const enum AVPixelFormat out_pix_fmts[] = { static const enum AVPixelFormat out_pix_fmts[] = {
AV_PIX_FMT_NV12, AV_PIX_FMT_NV12,
@@ -269,7 +269,21 @@ Index: FFmpeg/libavfilter/vf_vpp_qsv.c
INIT_MFX_EXTBUF(scale_conf, MFX_EXTBUFF_VPP_SCALING); INIT_MFX_EXTBUF(scale_conf, MFX_EXTBUFF_VPP_SCALING);
SET_MFX_PARAM_FIELD(scale_conf, ScalingMode, mode); SET_MFX_PARAM_FIELD(scale_conf, ScalingMode, mode);
@@ -884,19 +924,13 @@ static const AVOption vpp_options[] = { @@ -706,6 +746,13 @@ static int config_output(AVFilterLink *o
#undef INIT_MFX_EXTBUF
#undef SET_MFX_PARAM_FIELD
+ if (vpp->tonemap) {
+ av_frame_side_data_remove(&outlink->side_data, &outlink->nb_side_data,
+ AV_FRAME_DATA_MASTERING_DISPLAY_METADATA);
+ av_frame_side_data_remove(&outlink->side_data, &outlink->nb_side_data,
+ AV_FRAME_DATA_CONTENT_LIGHT_LEVEL);
+ }
+
if (vpp->use_frc || vpp->use_crop || vpp->deinterlace || vpp->denoise ||
vpp->detail || vpp->procamp || vpp->rotate || vpp->hflip ||
inlink->w != outlink->w || inlink->h != outlink->h || in_format != vpp->out_format ||
@@ -884,19 +931,13 @@ static const AVOption vpp_options[] = {
{ "height", "Output video height(0=input video height, -1=keep input video aspect)", OFFSET(oh), AV_OPT_TYPE_STRING, { .str="w*ch/cw" }, 0, 255, .flags = FLAGS }, { "height", "Output video height(0=input video height, -1=keep input video aspect)", OFFSET(oh), AV_OPT_TYPE_STRING, { .str="w*ch/cw" }, 0, 255, .flags = FLAGS },
{ "format", "Output pixel format", OFFSET(output_format_str), AV_OPT_TYPE_STRING, { .str = "same" }, .flags = FLAGS }, { "format", "Output pixel format", OFFSET(output_format_str), AV_OPT_TYPE_STRING, { .str = "same" }, .flags = FLAGS },
{ "async_depth", "Internal parallelization depth, the higher the value the higher the latency.", OFFSET(qsv.async_depth), AV_OPT_TYPE_INT, { .i64 = 4 }, 0, INT_MAX, .flags = FLAGS }, { "async_depth", "Internal parallelization depth, the higher the value the higher the latency.", OFFSET(qsv.async_depth), AV_OPT_TYPE_INT, { .i64 = 4 }, 0, INT_MAX, .flags = FLAGS },
@@ -290,7 +304,7 @@ Index: FFmpeg/libavfilter/vf_vpp_qsv.c
{ "rate", "Generate output at frame rate or field rate, available only for deinterlace mode", { "rate", "Generate output at frame rate or field rate, available only for deinterlace mode",
OFFSET(field_rate), AV_OPT_TYPE_INT, { .i64 = 0 }, 0, 1, FLAGS, .unit = "rate" }, OFFSET(field_rate), AV_OPT_TYPE_INT, { .i64 = 0 }, 0, 1, FLAGS, .unit = "rate" },
@@ -927,8 +961,9 @@ static const AVOption vpp_options[] = { @@ -927,8 +968,9 @@ static const AVOption vpp_options[] = {
{ "out_color_transfer", "Output color transfer characteristics", { "out_color_transfer", "Output color transfer characteristics",
OFFSET(color_transfer_str), AV_OPT_TYPE_STRING, { .str = NULL }, .flags = FLAGS }, OFFSET(color_transfer_str), AV_OPT_TYPE_STRING, { .str = NULL }, .flags = FLAGS },
@@ -301,7 +315,7 @@ Index: FFmpeg/libavfilter/vf_vpp_qsv.c
{ NULL } { NULL }
}; };
@@ -982,19 +1017,14 @@ static const AVOption qsvscale_options[] @@ -984,19 +1026,14 @@ static const AVOption qsvscale_options[]
{ "h", "Output video height(0=input video height, -1=keep input video aspect)", OFFSET(oh), AV_OPT_TYPE_STRING, { .str = "ih" }, .flags = FLAGS }, { "h", "Output video height(0=input video height, -1=keep input video aspect)", OFFSET(oh), AV_OPT_TYPE_STRING, { .str = "ih" }, .flags = FLAGS },
{ "format", "Output pixel format", OFFSET(output_format_str), AV_OPT_TYPE_STRING, { .str = "same" }, .flags = FLAGS }, { "format", "Output pixel format", OFFSET(output_format_str), AV_OPT_TYPE_STRING, { .str = "same" }, .flags = FLAGS },
@@ -323,7 +337,7 @@ Index: FFmpeg/libavfilter/vf_vpp_qsv.c
{ NULL }, { NULL },
}; };
@@ -1019,6 +1049,7 @@ static const AVOption qsvdeint_options[] @@ -1021,6 +1058,7 @@ static const AVOption qsvdeint_options[]
{ "bob", "bob algorithm", 0, AV_OPT_TYPE_CONST, {.i64 = MFX_DEINTERLACING_BOB}, MFX_DEINTERLACING_BOB, MFX_DEINTERLACING_ADVANCED, FLAGS, .unit = "mode"}, { "bob", "bob algorithm", 0, AV_OPT_TYPE_CONST, {.i64 = MFX_DEINTERLACING_BOB}, MFX_DEINTERLACING_BOB, MFX_DEINTERLACING_ADVANCED, FLAGS, .unit = "mode"},
{ "advanced", "Motion adaptive algorithm", 0, AV_OPT_TYPE_CONST, {.i64 = MFX_DEINTERLACING_ADVANCED}, MFX_DEINTERLACING_BOB, MFX_DEINTERLACING_ADVANCED, FLAGS, .unit = "mode"}, { "advanced", "Motion adaptive algorithm", 0, AV_OPT_TYPE_CONST, {.i64 = MFX_DEINTERLACING_ADVANCED}, MFX_DEINTERLACING_BOB, MFX_DEINTERLACING_ADVANCED, FLAGS, .unit = "mode"},
@@ -2,12 +2,12 @@ Index: FFmpeg/libavfilter/buffersrc.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/buffersrc.c --- FFmpeg.orig/libavfilter/buffersrc.c
+++ FFmpeg/libavfilter/buffersrc.c +++ FFmpeg/libavfilter/buffersrc.c
@@ -74,7 +74,7 @@ typedef struct BufferSourceContext { @@ -76,7 +76,7 @@ typedef struct BufferSourceContext {
c->prev_delta = c->prev_w != width || c->prev_h != height || c->prev_pix_fmt != format ||\ c->prev_delta = c->prev_w != width || c->prev_h != height || c->prev_pix_fmt != format ||\
c->prev_color_space != csp || c->prev_color_range != range;\ c->prev_color_space != csp || c->prev_color_range != range || c->prev_alpha_mode != alpha;\
if (c->link_delta) {\ if (c->link_delta) {\
- int loglevel = c->prev_delta ? AV_LOG_WARNING : AV_LOG_DEBUG;\ - int loglevel = c->prev_delta ? AV_LOG_WARNING : AV_LOG_DEBUG;\
+ int loglevel = c->prev_delta ? AV_LOG_VERBOSE : AV_LOG_DEBUG;\ + int loglevel = c->prev_delta ? AV_LOG_VERBOSE : AV_LOG_DEBUG;\
av_log(s, loglevel, "Changing video frame properties on the fly is not supported by all filters.\n");\ av_log(s, loglevel, "Changing video frame properties on the fly is not supported by all filters.\n");\
av_log(s, loglevel, "filter context - w: %d h: %d fmt: %d csp: %s range: %s, incoming frame - w: %d h: %d fmt: %d csp: %s range: %s pts_time: %s\n",\ av_log(s, loglevel, "filter context - w: %d h: %d fmt: %d csp: %s range: %s alpha: %s, incoming frame - w: %d h: %d fmt: %d csp: %s range: %s alpha: %s pts_time: %s\n",\
c->w, c->h, c->pix_fmt, av_color_space_name(c->color_space), av_color_range_name(c->color_range),\ c->w, c->h, c->pix_fmt, av_color_space_name(c->color_space), av_color_range_name(c->color_range), av_alpha_mode_name(c->alpha_mode),\
@@ -2,16 +2,16 @@ Index: FFmpeg/libavfilter/drawutils.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/drawutils.c --- FFmpeg.orig/libavfilter/drawutils.c
+++ FFmpeg/libavfilter/drawutils.c +++ FFmpeg/libavfilter/drawutils.c
@@ -437,7 +437,7 @@ void ff_blend_rectangle(FFDrawContext *d @@ -461,7 +461,7 @@ void ff_blend_rectangle(FFDrawContext *d
static void blend_pixel16(uint8_t *dst, unsigned src, unsigned alpha, static void blend_pixel16(uint8_t *dst, unsigned src, unsigned alpha,
const uint8_t *mask, int mask_linesize, int l2depth, const uint8_t *mask, int mask_linesize, int l2depth,
- unsigned w, unsigned h, unsigned shift, unsigned xm0) - unsigned w, unsigned h, unsigned shift, unsigned xm0)
+ unsigned w, unsigned h, unsigned shift, unsigned xm0, int srca_opaque) + unsigned w, unsigned h, unsigned shift, unsigned xm0, int srca_opaque)
{ {
unsigned xm, x, y, t = 0; unsigned t = 0;
unsigned xmshf = 3 - l2depth; unsigned xmshf = 3 - l2depth;
@@ -456,12 +456,13 @@ static void blend_pixel16(uint8_t *dst, @@ -480,12 +480,13 @@ static void blend_pixel16(uint8_t *dst,
mask += mask_linesize; mask += mask_linesize;
} }
alpha = (t >> shift) * alpha; alpha = (t >> shift) * alpha;
@@ -24,9 +24,9 @@ Index: FFmpeg/libavfilter/drawutils.c
- unsigned w, unsigned h, unsigned shift, unsigned xm0) - unsigned w, unsigned h, unsigned shift, unsigned xm0)
+ unsigned w, unsigned h, unsigned shift, unsigned xm0, int srca_opaque) + unsigned w, unsigned h, unsigned shift, unsigned xm0, int srca_opaque)
{ {
unsigned xm, x, y, t = 0; unsigned t = 0;
unsigned xmshf = 3 - l2depth; unsigned xmshf = 3 - l2depth;
@@ -479,6 +480,7 @@ static void blend_pixel(uint8_t *dst, un @@ -503,6 +504,7 @@ static void blend_pixel(uint8_t *dst, un
mask += mask_linesize; mask += mask_linesize;
} }
alpha = (t >> shift) * alpha; alpha = (t >> shift) * alpha;
@@ -34,14 +34,13 @@ Index: FFmpeg/libavfilter/drawutils.c
*dst = ((0x1010101 - alpha) * *dst + alpha * src) >> 24; *dst = ((0x1010101 - alpha) * *dst + alpha * src) >> 24;
} }
@@ -486,50 +488,134 @@ static void blend_line_hv16(uint8_t *dst @@ -510,48 +512,130 @@ static void blend_line_hv16(uint8_t *dst
unsigned src, unsigned alpha, unsigned src, unsigned alpha,
const uint8_t *mask, int mask_linesize, int l2depth, int w, const uint8_t *mask, int mask_linesize, int l2depth, int w,
unsigned hsub, unsigned vsub, unsigned hsub, unsigned vsub,
- int xm, int left, int right, int hband) - int xm, int left, int right, int hband)
+ int xm, int left, int right, int hband, int srca_opaque) + int xm, int left, int right, int hband, int srca_opaque)
{ {
int x;
if (left) { if (left) {
blend_pixel16(dst, src, alpha, mask, mask_linesize, l2depth, blend_pixel16(dst, src, alpha, mask, mask_linesize, l2depth,
@@ -50,7 +49,7 @@ Index: FFmpeg/libavfilter/drawutils.c
dst += dst_delta; dst += dst_delta;
xm += left; xm += left;
} }
for (x = 0; x < w; x++) { for (int x = 0; x < w; x++) {
blend_pixel16(dst, src, alpha, mask, mask_linesize, l2depth, blend_pixel16(dst, src, alpha, mask, mask_linesize, l2depth,
- 1 << hsub, hband, hsub + vsub, xm); - 1 << hsub, hband, hsub + vsub, xm);
+ 1 << hsub, hband, hsub + vsub, xm, srca_opaque); + 1 << hsub, hband, hsub + vsub, xm, srca_opaque);
@@ -70,7 +69,6 @@ Index: FFmpeg/libavfilter/drawutils.c
- int xm, int left, int right, int hband) - int xm, int left, int right, int hband)
+ int xm, int left, int right, int hband, int srca_opaque) + int xm, int left, int right, int hband, int srca_opaque)
{ {
int x;
if (left) { if (left) {
blend_pixel(dst, src, alpha, mask, mask_linesize, l2depth, blend_pixel(dst, src, alpha, mask, mask_linesize, l2depth,
@@ -79,7 +77,7 @@ Index: FFmpeg/libavfilter/drawutils.c
dst += dst_delta; dst += dst_delta;
xm += left; xm += left;
} }
for (x = 0; x < w; x++) { for (int x = 0; x < w; x++) {
blend_pixel(dst, src, alpha, mask, mask_linesize, l2depth, blend_pixel(dst, src, alpha, mask, mask_linesize, l2depth,
- 1 << hsub, hband, hsub + vsub, xm); - 1 << hsub, hband, hsub + vsub, xm);
+ 1 << hsub, hband, hsub + vsub, xm, srca_opaque); + 1 << hsub, hband, hsub + vsub, xm, srca_opaque);
@@ -99,16 +97,16 @@ Index: FFmpeg/libavfilter/drawutils.c
+ unsigned w, unsigned h, unsigned shift, unsigned xm0, + unsigned w, unsigned h, unsigned shift, unsigned xm0,
+ int srca_opaque, int limited) + int srca_opaque, int limited)
+{ +{
+ unsigned xm, x, y, t = 0; + unsigned t = 0;
+ unsigned xmshf = 3 - l2depth; + unsigned xmshf = 3 - l2depth;
+ unsigned xmmod = 7 >> l2depth; + unsigned xmmod = 7 >> l2depth;
+ unsigned mbits = (1 << (1 << l2depth)) - 1; + unsigned mbits = (1 << (1 << l2depth)) - 1;
+ unsigned mmult = 255 / mbits; + unsigned mmult = 255 / mbits;
+ const uint8_t offset = limited ? 16 : 0; + const uint8_t offset = limited ? 16 : 0;
+ +
+ for (y = 0; y < h; y++) { + for (unsigned y = 0; y < h; y++) {
+ xm = xm0; + unsigned xm = xm0;
+ for (x = 0; x < w; x++) { + for (unsigned x = 0; x < w; x++) {
+ t += ((mask[xm >> xmshf] >> ((~xm & xmmod) << l2depth)) & mbits) + t += ((mask[xm >> xmshf] >> ((~xm & xmmod) << l2depth)) & mbits)
+ * mmult; + * mmult;
+ xm++; + xm++;
@@ -145,8 +143,6 @@ Index: FFmpeg/libavfilter/drawutils.c
+ int xm, int left, int right, int hband, + int xm, int left, int right, int hband,
+ int srca_opaque, int limited) + int srca_opaque, int limited)
+{ +{
+ int x;
+
+ if (left) { + if (left) {
+ blend_pixel_unpremul_rgb32( + blend_pixel_unpremul_rgb32(
+ dst0, dst1, dst2, dst3, + dst0, dst1, dst2, dst3,
@@ -158,7 +154,7 @@ Index: FFmpeg/libavfilter/drawutils.c
+ dst3 += dst_delta; + dst3 += dst_delta;
+ xm += left; + xm += left;
+ } + }
+ for (x = 0; x < w; x++) { + for (int x = 0; x < w; x++) {
+ blend_pixel_unpremul_rgb32( + blend_pixel_unpremul_rgb32(
+ dst0, dst1, dst2, dst3, + dst0, dst1, dst2, dst3,
+ src, alpha, mask, mask_linesize, l2depth, + src, alpha, mask, mask_linesize, l2depth,
@@ -177,7 +173,7 @@ Index: FFmpeg/libavfilter/drawutils.c
} }
void ff_blend_mask(FFDrawContext *draw, FFDrawColor *color, void ff_blend_mask(FFDrawContext *draw, FFDrawColor *color,
@@ -567,13 +653,79 @@ void ff_blend_mask(FFDrawContext *draw, @@ -589,13 +673,79 @@ void ff_blend_mask(FFDrawContext *draw,
y_sub = y0; y_sub = y0;
subsampling_bounds(draw->hsub[plane], &x_sub, &w_sub, &left, &right); subsampling_bounds(draw->hsub[plane], &x_sub, &w_sub, &left, &right);
subsampling_bounds(draw->vsub[plane], &y_sub, &h_sub, &top, &bottom); subsampling_bounds(draw->vsub[plane], &y_sub, &h_sub, &top, &bottom);
@@ -195,7 +191,7 @@ Index: FFmpeg/libavfilter/drawutils.c
+ uint8_t *pdst[4]; + uint8_t *pdst[4];
+ unsigned src[4]; + unsigned src[4];
+ +
+ for (comp = 0; comp < nb_comp; comp++) { + for (unsigned comp = 0; comp < nb_comp; comp++) {
+ av_assert0(draw->desc->comp[comp].plane == plane); + av_assert0(draw->desc->comp[comp].plane == plane);
+ +
+ offset[comp] = draw->desc->comp[comp].offset; + offset[comp] = draw->desc->comp[comp].offset;
@@ -213,11 +209,11 @@ Index: FFmpeg/libavfilter/drawutils.c
+ draw->hsub[plane], draw->vsub[plane], + draw->hsub[plane], draw->vsub[plane],
+ xm0, left, right, top, srca_opaque, limited); + xm0, left, right, top, srca_opaque, limited);
+ +
+ for (comp = 0; comp < nb_comp; comp++) + for (unsigned comp = 0; comp < nb_comp; comp++)
+ pdst[comp] += dst_linesize[plane]; + pdst[comp] += dst_linesize[plane];
+ m += top * mask_linesize; + m += top * mask_linesize;
+ } + }
+ for (y = 0; y < h_sub; y++) { + for (int y = 0; y < h_sub; y++) {
+ blend_line_hv_unpremul_rgb32( + blend_line_hv_unpremul_rgb32(
+ pdst[0], pdst[1], pdst[2], pdst[3], draw->pixelstep[plane], + pdst[0], pdst[1], pdst[2], pdst[3], draw->pixelstep[plane],
+ src, alpha, + src, alpha,
@@ -225,7 +221,7 @@ Index: FFmpeg/libavfilter/drawutils.c
+ draw->hsub[plane], draw->vsub[plane], + draw->hsub[plane], draw->vsub[plane],
+ xm0, left, right, 1 << draw->vsub[plane], srca_opaque, limited); + xm0, left, right, 1 << draw->vsub[plane], srca_opaque, limited);
+ +
+ for (comp = 0; comp < nb_comp; comp++) + for (unsigned comp = 0; comp < nb_comp; comp++)
+ pdst[comp] += dst_linesize[plane]; + pdst[comp] += dst_linesize[plane];
+ m += mask_linesize << draw->vsub[plane]; + m += mask_linesize << draw->vsub[plane];
+ } + }
@@ -240,7 +236,7 @@ Index: FFmpeg/libavfilter/drawutils.c
+ break; + break;
+ } + }
+ +
for (comp = 0; comp < nb_comp; comp++) { for (unsigned comp = 0; comp < nb_comp; comp++) {
const int depth = draw->desc->comp[comp].depth; const int depth = draw->desc->comp[comp].depth;
const int offset = draw->desc->comp[comp].offset; const int offset = draw->desc->comp[comp].offset;
const int index = offset / ((depth + 7) / 8); const int index = offset / ((depth + 7) / 8);
@@ -257,7 +253,7 @@ Index: FFmpeg/libavfilter/drawutils.c
p = p0 + offset; p = p0 + offset;
m = mask; m = mask;
if (top) { if (top) {
@@ -582,13 +734,13 @@ void ff_blend_mask(FFDrawContext *draw, @@ -604,13 +754,13 @@ void ff_blend_mask(FFDrawContext *draw,
color->comp[plane].u8[index], alpha, color->comp[plane].u8[index], alpha,
m, mask_linesize, l2depth, w_sub, m, mask_linesize, l2depth, w_sub,
draw->hsub[plane], draw->vsub[plane], draw->hsub[plane], draw->vsub[plane],
@@ -273,7 +269,7 @@ Index: FFmpeg/libavfilter/drawutils.c
} }
p += dst_linesize[plane]; p += dst_linesize[plane];
m += top * mask_linesize; m += top * mask_linesize;
@@ -599,7 +751,7 @@ void ff_blend_mask(FFDrawContext *draw, @@ -621,7 +771,7 @@ void ff_blend_mask(FFDrawContext *draw,
color->comp[plane].u8[index], alpha, color->comp[plane].u8[index], alpha,
m, mask_linesize, l2depth, w_sub, m, mask_linesize, l2depth, w_sub,
draw->hsub[plane], draw->vsub[plane], draw->hsub[plane], draw->vsub[plane],
@@ -282,7 +278,7 @@ Index: FFmpeg/libavfilter/drawutils.c
p += dst_linesize[plane]; p += dst_linesize[plane];
m += mask_linesize << draw->vsub[plane]; m += mask_linesize << draw->vsub[plane];
} }
@@ -609,7 +761,7 @@ void ff_blend_mask(FFDrawContext *draw, @@ -631,7 +781,7 @@ void ff_blend_mask(FFDrawContext *draw,
color->comp[plane].u16[index], alpha, color->comp[plane].u16[index], alpha,
m, mask_linesize, l2depth, w_sub, m, mask_linesize, l2depth, w_sub,
draw->hsub[plane], draw->vsub[plane], draw->hsub[plane], draw->vsub[plane],
@@ -291,7 +287,7 @@ Index: FFmpeg/libavfilter/drawutils.c
p += dst_linesize[plane]; p += dst_linesize[plane];
m += mask_linesize << draw->vsub[plane]; m += mask_linesize << draw->vsub[plane];
} }
@@ -620,13 +772,13 @@ void ff_blend_mask(FFDrawContext *draw, @@ -642,13 +792,13 @@ void ff_blend_mask(FFDrawContext *draw,
color->comp[plane].u8[index], alpha, color->comp[plane].u8[index], alpha,
m, mask_linesize, l2depth, w_sub, m, mask_linesize, l2depth, w_sub,
draw->hsub[plane], draw->vsub[plane], draw->hsub[plane], draw->vsub[plane],
@@ -307,16 +303,11 @@ Index: FFmpeg/libavfilter/drawutils.c
} }
} }
} }
@@ -658,3 +810,4 @@ AVFilterFormats *ff_draw_supported_pixel
return NULL;
return fmts;
}
+
Index: FFmpeg/libavfilter/drawutils.h Index: FFmpeg/libavfilter/drawutils.h
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/drawutils.h --- FFmpeg.orig/libavfilter/drawutils.h
+++ FFmpeg/libavfilter/drawutils.h +++ FFmpeg/libavfilter/drawutils.h
@@ -59,7 +59,17 @@ typedef struct FFDrawColor { @@ -61,7 +61,17 @@ typedef struct FFDrawColor {
/** /**
* Process alpha pixel component. * Process alpha pixel component.
*/ */
@@ -335,11 +326,6 @@ Index: FFmpeg/libavfilter/drawutils.h
/** /**
* Init a draw context. * Init a draw context.
@@ -164,3 +174,4 @@ int ff_draw_round_to_sub(FFDrawContext *
AVFilterFormats *ff_draw_supported_pixel_formats(unsigned flags);
#endif /* AVFILTER_DRAWUTILS_H */
+
Index: FFmpeg/libavfilter/vf_subtitles.c Index: FFmpeg/libavfilter/vf_subtitles.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/vf_subtitles.c --- FFmpeg.orig/libavfilter/vf_subtitles.c
@@ -376,19 +362,15 @@ Index: FFmpeg/libavfilter/vf_subtitles.c
/* libass supports a log level ranging from 0 to 7 */ /* libass supports a log level ranging from 0 to 7 */
static const int ass_libavfilter_log_level_map[] = { static const int ass_libavfilter_log_level_map[] = {
@@ -146,9 +155,25 @@ static int query_formats(AVFilterContext @@ -184,12 +193,28 @@ static int config_input(AVFilterLink *in
static int config_input(AVFilterLink *inlink)
{ {
AssContext *ass = inlink->dst->priv; AVFilterContext *ctx = inlink->dst;
AssContext *ass = ctx->priv;
+ const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(inlink->format);
+ int draw_flags = ass->alpha ? FF_DRAW_PROCESS_ALPHA : 0; + int draw_flags = ass->alpha ? FF_DRAW_PROCESS_ALPHA : 0;
int ret;
- ff_draw_init2(&ass->draw, inlink->format, inlink->colorspace, inlink->color_range, + if (ass->sub2video && inlink->alpha_mode != AVALPHA_MODE_PREMULTIPLIED) {
- ass->alpha ? FF_DRAW_PROCESS_ALPHA : 0);
+ if (ass->sub2video) {
+ const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(inlink->format);
+ if (!desc)
+ return AVERROR(ENOSYS);
+
+ if ((desc->flags & AV_PIX_FMT_FLAG_ALPHA) && + if ((desc->flags & AV_PIX_FMT_FLAG_ALPHA) &&
+ (desc->flags & (AV_PIX_FMT_FLAG_PLANAR | AV_PIX_FMT_FLAG_RGB))) + (desc->flags & (AV_PIX_FMT_FLAG_PLANAR | AV_PIX_FMT_FLAG_RGB)))
+ draw_flags |= FF_DRAW_MASK_SRC_ALPHA_OPAQUE | FF_DRAW_PROCESS_ALPHA; + draw_flags |= FF_DRAW_MASK_SRC_ALPHA_OPAQUE | FF_DRAW_PROCESS_ALPHA;
@@ -400,11 +382,17 @@ Index: FFmpeg/libavfilter/vf_subtitles.c
+ draw_flags |= FF_DRAW_MASK_UNPREMUL_RGB32; + draw_flags |= FF_DRAW_MASK_UNPREMUL_RGB32;
+ } + }
+ +
+ ff_draw_init2(&ass->draw, inlink->format, inlink->colorspace, inlink->color_range, draw_flags); ret = ff_draw_init2(&ass->draw, inlink->format,
+ (desc->flags & AV_PIX_FMT_FLAG_RGB) ? inlink->colorspace :
ass_set_frame_size (ass->renderer, inlink->w, inlink->h); ass_get_color_space(ass->track->YCbCrMatrix, inlink->colorspace),
if (ass->original_w && ass->original_h) { + (desc->flags & AV_PIX_FMT_FLAG_RGB) ? inlink->color_range :
@@ -161,6 +186,8 @@ static int config_input(AVFilterLink *in ass_get_color_range(ass->track->YCbCrMatrix, inlink->color_range),
- inlink->alpha_mode, ass->alpha ? FF_DRAW_PROCESS_ALPHA : 0);
+ inlink->alpha_mode, draw_flags);
if (ret < 0) {
av_log(ctx, AV_LOG_ERROR, "Failed to initialize FFDrawContext\n");
return ret;
@@ -206,6 +231,8 @@ static int config_input(AVFilterLink *in
if (ass->shaping != -1) if (ass->shaping != -1)
ass_set_shaper(ass->renderer, ass->shaping); ass_set_shaper(ass->renderer, ass->shaping);
@@ -413,7 +401,7 @@ Index: FFmpeg/libavfilter/vf_subtitles.c
return 0; return 0;
} }
@@ -189,20 +216,44 @@ static int filter_frame(AVFilterLink *in @@ -234,20 +261,44 @@ static int filter_frame(AVFilterLink *in
{ {
AVFilterContext *ctx = inlink->dst; AVFilterContext *ctx = inlink->dst;
AVFilterLink *outlink = ctx->outputs[0]; AVFilterLink *outlink = ctx->outputs[0];
@@ -460,7 +448,7 @@ Index: FFmpeg/libavfilter/vf_subtitles.c
static const AVFilterPad ass_inputs[] = { static const AVFilterPad ass_inputs[] = {
{ {
.name = "default", .name = "default",
@@ -217,10 +268,6 @@ static const AVFilterPad ass_inputs[] = @@ -262,10 +313,6 @@ static const AVFilterPad ass_inputs[] =
static const AVOption ass_options[] = { static const AVOption ass_options[] = {
COMMON_OPTIONS COMMON_OPTIONS
@@ -471,7 +459,7 @@ Index: FFmpeg/libavfilter/vf_subtitles.c
{NULL}, {NULL},
}; };
@@ -244,6 +291,9 @@ static av_cold int init_ass(AVFilterCont @@ -289,6 +336,9 @@ static av_cold int init_ass(AVFilterCont
ass->filename); ass->filename);
return AVERROR(EINVAL); return AVERROR(EINVAL);
} }
@@ -481,7 +469,7 @@ Index: FFmpeg/libavfilter/vf_subtitles.c
return 0; return 0;
} }
@@ -265,8 +315,8 @@ const AVFilter ff_vf_ass = { @@ -310,8 +360,8 @@ const FFFilter ff_vf_ass = {
static const AVOption subtitles_options[] = { static const AVOption subtitles_options[] = {
COMMON_OPTIONS COMMON_OPTIONS
{"charenc", "set input character encoding", OFFSET(charenc), AV_OPT_TYPE_STRING, {.str = NULL}, 0, 0, FLAGS}, {"charenc", "set input character encoding", OFFSET(charenc), AV_OPT_TYPE_STRING, {.str = NULL}, 0, 0, FLAGS},
@@ -492,7 +480,7 @@ Index: FFmpeg/libavfilter/vf_subtitles.c
{"force_style", "force subtitle style", OFFSET(force_style), AV_OPT_TYPE_STRING, {.str = NULL}, 0, 0, FLAGS}, {"force_style", "force subtitle style", OFFSET(force_style), AV_OPT_TYPE_STRING, {.str = NULL}, 0, 0, FLAGS},
#if FF_ASS_FEATURE_WRAP_UNICODE #if FF_ASS_FEATURE_WRAP_UNICODE
{"wrap_unicode", "break lines according to the Unicode Line Breaking Algorithm", OFFSET(wrap_unicode), AV_OPT_TYPE_BOOL, { .i64 = -1 }, -1, 1, FLAGS }, {"wrap_unicode", "break lines according to the Unicode Line Breaking Algorithm", OFFSET(wrap_unicode), AV_OPT_TYPE_BOOL, { .i64 = -1 }, -1, 1, FLAGS },
@@ -495,6 +545,8 @@ static av_cold int init_subtitles(AVFilt @@ -540,6 +590,8 @@ static av_cold int init_subtitles(AVFilt
avsubtitle_free(&sub); avsubtitle_free(&sub);
} }
@@ -2,7 +2,7 @@ Index: FFmpeg/libavfilter/Makefile
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/Makefile --- FFmpeg.orig/libavfilter/Makefile
+++ FFmpeg/libavfilter/Makefile +++ FFmpeg/libavfilter/Makefile
@@ -593,6 +593,7 @@ OBJS-$(CONFIG_DRAWBOX_VAAPI_FILTER) @@ -611,6 +611,7 @@ OBJS-$(CONFIG_DRAWBOX_VAAPI_FILTER)
OBJS-$(CONFIG_ALLRGB_FILTER) += vsrc_testsrc.o OBJS-$(CONFIG_ALLRGB_FILTER) += vsrc_testsrc.o
OBJS-$(CONFIG_ALLYUV_FILTER) += vsrc_testsrc.o OBJS-$(CONFIG_ALLYUV_FILTER) += vsrc_testsrc.o
@@ -14,19 +14,19 @@ Index: FFmpeg/libavfilter/allfilters.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/allfilters.c --- FFmpeg.orig/libavfilter/allfilters.c
+++ FFmpeg/libavfilter/allfilters.c +++ FFmpeg/libavfilter/allfilters.c
@@ -556,6 +556,7 @@ extern const AVFilter ff_vf_drawbox_vaap @@ -571,6 +571,7 @@ extern const FFFilter ff_vf_drawbox_vaap
extern const AVFilter ff_vsrc_allrgb; extern const FFFilter ff_vsrc_allrgb;
extern const AVFilter ff_vsrc_allyuv; extern const FFFilter ff_vsrc_allyuv;
+extern const AVFilter ff_vsrc_alphasrc; +extern const FFFilter ff_vsrc_alphasrc;
extern const AVFilter ff_vsrc_cellauto; extern const FFFilter ff_vsrc_amf_capture;
extern const AVFilter ff_vsrc_color; extern const FFFilter ff_vsrc_cellauto;
extern const AVFilter ff_vsrc_color_vulkan; extern const FFFilter ff_vsrc_color;
Index: FFmpeg/libavfilter/vsrc_alphasrc.c Index: FFmpeg/libavfilter/vsrc_alphasrc.c
=================================================================== ===================================================================
--- /dev/null --- /dev/null
+++ FFmpeg/libavfilter/vsrc_alphasrc.c +++ FFmpeg/libavfilter/vsrc_alphasrc.c
@@ -0,0 +1,165 @@ @@ -0,0 +1,160 @@
+/* +/*
+ * Copyright (c) 2021 NyanMisaka + * Copyright (c) 2021 NyanMisaka
+ * + *
@@ -84,17 +84,12 @@ Index: FFmpeg/libavfilter/vsrc_alphasrc.c
+ return 0; + return 0;
+} +}
+ +
+static int alphasrc_query_formats(AVFilterContext *ctx) +static int alphasrc_query_formats(const AVFilterContext *ctx,
+ AVFilterFormatsConfig **cfg_in,
+ AVFilterFormatsConfig **cfg_out)
+{ +{
+ AVFilterLink *outlink = ctx->outputs[0]; + return ff_set_common_formats2(ctx, cfg_in, cfg_out,
+ AVFilterFormats *formats = ff_formats_pixdesc_filter(AV_PIX_FMT_FLAG_ALPHA, 0); + ff_formats_pixdesc_filter(AV_PIX_FMT_FLAG_ALPHA, 0));
+ int ret;
+
+ ret = ff_formats_ref(formats, &outlink->incfg.formats);
+ if (ret < 0)
+ return ret;
+
+ return 0;
+} +}
+ +
+static int alphasrc_config_output(AVFilterLink *outlink) +static int alphasrc_config_output(AVFilterLink *outlink)
@@ -181,14 +176,14 @@ Index: FFmpeg/libavfilter/vsrc_alphasrc.c
+ }, + },
+}; +};
+ +
+const AVFilter ff_vsrc_alphasrc = { +const FFFilter ff_vsrc_alphasrc = {
+ .name = "alphasrc", + .p.name = "alphasrc",
+ .description = NULL_IF_CONFIG_SMALL("Provide a blank video input with alpha channel."), + .p.description = NULL_IF_CONFIG_SMALL("Provide a blank video input with alpha channel."),
+ .priv_size = sizeof(AlphaSrc), + .priv_size = sizeof(AlphaSrc),
+ .priv_class = &alphasrc_class, + .p.priv_class = &alphasrc_class,
+ .init = alphasrc_init, + .init = alphasrc_init,
+ .uninit = NULL, + .uninit = NULL,
+ .inputs = NULL, + .p.inputs = NULL,
+ FILTER_OUTPUTS(alphasrc_outputs), + FILTER_OUTPUTS(alphasrc_outputs),
+ FILTER_QUERY_FUNC(alphasrc_query_formats), + FILTER_QUERY_FUNC2(alphasrc_query_formats),
+}; +};
@@ -1,41 +0,0 @@
Index: FFmpeg/libavcodec/bsf/hevc_mp4toannexb.c
===================================================================
--- FFmpeg.orig/libavcodec/bsf/hevc_mp4toannexb.c
+++ FFmpeg/libavcodec/bsf/hevc_mp4toannexb.c
@@ -127,6 +127,7 @@ static int hevc_mp4toannexb_filter(AVBSF
int got_irap = 0;
int i, ret = 0;
+ int has_sps = 0, has_pps = 0;
ret = ff_bsf_get_packet(ctx, &in);
if (ret < 0)
@@ -158,11 +159,14 @@ static int hevc_mp4toannexb_filter(AVBSF
}
nalu_type = (bytestream2_peek_byte(&gb) >> 1) & 0x3f;
+ has_sps = (has_sps || nalu_type == HEVC_NAL_SPS);
+ has_pps = (has_pps || nalu_type == HEVC_NAL_PPS);
/* prepend extradata to IRAP frames */
is_irap = nalu_type >= HEVC_NAL_BLA_W_LP &&
nalu_type <= HEVC_NAL_RSV_IRAP_VCL23;
- add_extradata = is_irap && !got_irap;
+ /* ignore the extradata if IRAP frame has sps and pps */
+ add_extradata = is_irap && !got_irap && !(has_sps && has_pps);
extra_size = add_extradata * ctx->par_out->extradata_size;
got_irap |= is_irap;
Index: FFmpeg/tests/fate/hevc.mak
===================================================================
--- FFmpeg.orig/tests/fate/hevc.mak
+++ FFmpeg/tests/fate/hevc.mak
@@ -233,7 +233,7 @@ FATE_HEVC-$(call ALLYES, HEVC_DEMUXER MO
fate-hevc-bsf-mp4toannexb: tests/data/hevc-mp4.mov
fate-hevc-bsf-mp4toannexb: CMD = md5 -i $(TARGET_PATH)/tests/data/hevc-mp4.mov -c:v copy -fflags +bitexact -f hevc
fate-hevc-bsf-mp4toannexb: CMP = oneline
-fate-hevc-bsf-mp4toannexb: REF = 73019329ed7f81c24f9af67c34c640c0
+fate-hevc-bsf-mp4toannexb: REF = 7d05a79c7a6665ae22c0043a4d83a811
fate-hevc-skiploopfilter: CMD = framemd5 -skip_loop_filter nokey -i $(TARGET_SAMPLES)/hevc-conformance/SAO_D_Samsung_5.bit -sws_flags bitexact
FATE_HEVC-$(call FRAMEMD5, HEVC, HEVC, HEVC_PARSER) += fate-hevc-skiploopfilter
@@ -2,15 +2,7 @@ Index: FFmpeg/libavformat/id3v2.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavformat/id3v2.c --- FFmpeg.orig/libavformat/id3v2.c
+++ FFmpeg/libavformat/id3v2.c +++ FFmpeg/libavformat/id3v2.c
@@ -137,6 +137,7 @@ const CodecMime ff_id3v2_mime_tags[] = { @@ -335,39 +335,54 @@ static void read_ttag(AVFormatContext *s
{ "image/png", AV_CODEC_ID_PNG },
{ "image/tiff", AV_CODEC_ID_TIFF },
{ "image/bmp", AV_CODEC_ID_BMP },
+ { "image/webp", AV_CODEC_ID_WEBP },
{ "JPG", AV_CODEC_ID_MJPEG }, /* ID3v2.2 */
{ "PNG", AV_CODEC_ID_PNG }, /* ID3v2.2 */
{ "", AV_CODEC_ID_NONE },
@@ -326,39 +327,54 @@ static void read_ttag(AVFormatContext *s
AVDictionary **metadata, const char *key) AVDictionary **metadata, const char *key)
{ {
uint8_t *dst; uint8_t *dst;
@@ -2,15 +2,15 @@ Index: FFmpeg/configure
=================================================================== ===================================================================
--- FFmpeg.orig/configure --- FFmpeg.orig/configure
+++ FFmpeg/configure +++ FFmpeg/configure
@@ -1876,7 +1876,6 @@ EXTERNAL_LIBRARY_GPL_LIST=" @@ -2003,7 +2003,6 @@ EXTERNAL_LIBRARY_GPL_LIST="
EXTERNAL_LIBRARY_NONFREE_LIST=" EXTERNAL_LIBRARY_NONFREE_LIST="
decklink decklink
- libfdk_aac - libfdk_aac
libtls libmpeghdec
" "
@@ -1916,6 +1915,7 @@ EXTERNAL_LIBRARY_LIST=" @@ -2044,6 +2043,7 @@ EXTERNAL_LIBRARY_LIST="
libcodec2 libcodec2
libdav1d libdav1d
libdc1394 libdc1394
@@ -1,16 +0,0 @@
Index: FFmpeg/libavcodec/hw_base_encode_h265.c
===================================================================
--- FFmpeg.orig/libavcodec/hw_base_encode_h265.c
+++ FFmpeg/libavcodec/hw_base_encode_h265.c
@@ -216,8 +216,9 @@ int ff_hw_base_encode_init_params_h265(F
sps->log2_min_luma_transform_block_size_minus2 = 0;
sps->log2_diff_max_min_luma_transform_block_size = 3;
// Full transform hierarchy allowed (2-5).
- sps->max_transform_hierarchy_depth_inter = 3;
- sps->max_transform_hierarchy_depth_intra = 3;
+ // Default to 2 based on Programmer's Reference Manuals of Intel graphics.
+ sps->max_transform_hierarchy_depth_inter = 2;
+ sps->max_transform_hierarchy_depth_intra = 2;
// AMP works.
sps->amp_enabled_flag = 1;
// SAO and temporal MVP do not work.
@@ -2,7 +2,7 @@ Index: FFmpeg/fftools/ffmpeg.c
=================================================================== ===================================================================
--- FFmpeg.orig/fftools/ffmpeg.c --- FFmpeg.orig/fftools/ffmpeg.c
+++ FFmpeg/fftools/ffmpeg.c +++ FFmpeg/fftools/ffmpeg.c
@@ -95,6 +95,9 @@ typedef struct BenchmarkTimeStamps { @@ -99,6 +99,9 @@ typedef struct BenchmarkTimeStamps {
static BenchmarkTimeStamps get_benchmark_time_stamps(void); static BenchmarkTimeStamps get_benchmark_time_stamps(void);
static int64_t getmaxrss(void); static int64_t getmaxrss(void);
@@ -12,7 +12,7 @@ Index: FFmpeg/fftools/ffmpeg.c
atomic_uint nb_output_dumped = 0; atomic_uint nb_output_dumped = 0;
@@ -113,6 +116,9 @@ int nb_filtergraphs; @@ -117,6 +120,9 @@ int nb_filtergraphs;
Decoder **decoders; Decoder **decoders;
int nb_decoders; int nb_decoders;
@@ -22,7 +22,7 @@ Index: FFmpeg/fftools/ffmpeg.c
#if HAVE_TERMIOS_H #if HAVE_TERMIOS_H
/* init terminal so that we can grab keys */ /* init terminal so that we can grab keys */
@@ -785,6 +791,20 @@ static void set_tty_echo(int on) @@ -818,6 +825,20 @@ static void set_tty_echo(int on)
#endif #endif
} }
@@ -43,7 +43,7 @@ Index: FFmpeg/fftools/ffmpeg.c
static int check_keyboard_interaction(int64_t cur_time) static int check_keyboard_interaction(int64_t cur_time)
{ {
int i, key; int i, key;
@@ -801,6 +821,11 @@ static int check_keyboard_interaction(in @@ -834,6 +855,11 @@ static int check_keyboard_interaction(in
} }
if (key == '+') av_log_set_level(av_log_get_level()+10); if (key == '+') av_log_set_level(av_log_get_level()+10);
if (key == '-') av_log_set_level(av_log_get_level()-10); if (key == '-') av_log_set_level(av_log_get_level()-10);
@@ -55,7 +55,7 @@ Index: FFmpeg/fftools/ffmpeg.c
if (key == 'c' || key == 'C'){ if (key == 'c' || key == 'C'){
char buf[4096], target[64], command[256], arg[256] = {0}; char buf[4096], target[64], command[256], arg[256] = {0};
double time; double time;
@@ -840,7 +865,9 @@ static int check_keyboard_interaction(in @@ -873,7 +899,9 @@ static int check_keyboard_interaction(in
"c Send command to first matching filter supporting it\n" "c Send command to first matching filter supporting it\n"
"C Send/Queue command to all matching filters\n" "C Send/Queue command to all matching filters\n"
"h dump packets/hex press to cycle through the 3 states\n" "h dump packets/hex press to cycle through the 3 states\n"
@@ -65,7 +65,7 @@ Index: FFmpeg/fftools/ffmpeg.c
"s Show QP histogram\n" "s Show QP histogram\n"
); );
} }
@@ -870,15 +897,20 @@ static int transcode(Scheduler *sch) @@ -903,15 +931,20 @@ static int transcode(Scheduler *sch)
timer_start = av_gettime_relative(); timer_start = av_gettime_relative();
while (!sch_wait(sch, stats_period, &transcode_ts)) { while (!sch_wait(sch, stats_period, &transcode_ts)) {
@@ -90,7 +90,7 @@ Index: FFmpeg/fftools/ffmpeg.c
/* dump report by using the output first video and audio streams */ /* dump report by using the output first video and audio streams */
print_report(0, timer_start, cur_time, transcode_ts); print_report(0, timer_start, cur_time, transcode_ts);
@@ -895,11 +927,17 @@ static int transcode(Scheduler *sch) @@ -928,11 +961,17 @@ static int transcode(Scheduler *sch)
term_exit(); term_exit();
/* dump report by using the first video and audio streams */ /* dump report by using the first video and audio streams */
@@ -113,7 +113,7 @@ Index: FFmpeg/fftools/ffmpeg.h
=================================================================== ===================================================================
--- FFmpeg.orig/fftools/ffmpeg.h --- FFmpeg.orig/fftools/ffmpeg.h
+++ FFmpeg/fftools/ffmpeg.h +++ FFmpeg/fftools/ffmpeg.h
@@ -748,6 +748,9 @@ extern int recast_media; @@ -802,6 +802,9 @@ extern int recast_media;
extern FILE *vstats_file; extern FILE *vstats_file;
@@ -127,7 +127,7 @@ Index: FFmpeg/fftools/ffmpeg_demux.c
=================================================================== ===================================================================
--- FFmpeg.orig/fftools/ffmpeg_demux.c --- FFmpeg.orig/fftools/ffmpeg_demux.c
+++ FFmpeg/fftools/ffmpeg_demux.c +++ FFmpeg/fftools/ffmpeg_demux.c
@@ -704,6 +704,11 @@ static int input_thread(void *arg) @@ -748,6 +748,11 @@ static int input_thread(void *arg)
DemuxStream *ds; DemuxStream *ds;
unsigned send_flags = 0; unsigned send_flags = 0;
@@ -2,7 +2,7 @@ Index: FFmpeg/configure
=================================================================== ===================================================================
--- FFmpeg.orig/configure --- FFmpeg.orig/configure
+++ FFmpeg/configure +++ FFmpeg/configure
@@ -5601,7 +5601,7 @@ elif enabled x86; then @@ -5914,7 +5914,7 @@ elif enabled x86; then
;; ;;
# everything else should support nopl and conditional mov (cmov) # everything else should support nopl and conditional mov (cmov)
*) *)
@@ -11,17 +11,17 @@ Index: FFmpeg/configure
enable i686 enable i686
enable fast_cmov enable fast_cmov
;; ;;
@@ -7592,7 +7592,9 @@ if enabled icc; then @@ -8052,8 +8052,8 @@ if enabled icc; then
disable aligned_stack
fi
elif enabled gcc; then elif enabled gcc; then
- check_optflags -fno-tree-vectorize gcc_version=$($cc -dumpversion)
+ case $gcc_basever in major_version=${gcc_version%%[!0-9]*}
+ 2|2.*|3.*|4.*|5.*|6.*) check_optflags -fno-tree-vectorize ;; - if [ $major_version -lt 13 ]; then
+ esac - # Disable tree-vectorize for GCC <13 - it has historically been buggy.
check_cflags -Werror=format-security + if [ $major_version -lt 11 ]; then
check_cflags -Werror=implicit-function-declaration + # Disable tree-vectorize for GCC <11 - it has historically been buggy.
check_cflags -Werror=missing-prototypes check_optflags -fno-tree-vectorize
else
case $arch in
Index: FFmpeg/libavcodec/huffyuvdsp.c Index: FFmpeg/libavcodec/huffyuvdsp.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavcodec/huffyuvdsp.c --- FFmpeg.orig/libavcodec/huffyuvdsp.c
@@ -66,7 +66,7 @@ Index: FFmpeg/libavcodec/x86/cabac.h
#else #else
av_always_inline av_always_inline
#endif #endif
+#ifdef __GNUC__ +#if defined(__GNUC__) && !defined(__clang__)
+__attribute__((optimize("-fno-tree-vectorize"))) +__attribute__((optimize("-fno-tree-vectorize")))
+#endif +#endif
int get_cabac_inline_x86(CABACContext *c, uint8_t *const state) int get_cabac_inline_x86(CABACContext *c, uint8_t *const state)
@@ -35,7 +35,7 @@ Index: FFmpeg/libavutil/hwcontext_d3d11va.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavutil/hwcontext_d3d11va.c --- FFmpeg.orig/libavutil/hwcontext_d3d11va.c
+++ FFmpeg/libavutil/hwcontext_d3d11va.c +++ FFmpeg/libavutil/hwcontext_d3d11va.c
@@ -611,6 +611,35 @@ static int d3d11va_device_find_adapter_b @@ -622,6 +622,35 @@ static int d3d11va_device_find_adapter_b
return -1; return -1;
} }
@@ -71,7 +71,7 @@ Index: FFmpeg/libavutil/hwcontext_d3d11va.c
static int d3d11va_device_create(AVHWDeviceContext *ctx, const char *device, static int d3d11va_device_create(AVHWDeviceContext *ctx, const char *device,
AVDictionary *opts, int flags) AVDictionary *opts, int flags)
{ {
@@ -688,6 +717,9 @@ static int d3d11va_device_create(AVHWDev @@ -699,6 +728,9 @@ static int d3d11va_device_create(AVHWDev
ID3D10Multithread_Release(pMultithread); ID3D10Multithread_Release(pMultithread);
} }
@@ -85,10 +85,10 @@ Index: FFmpeg/libavutil/hwcontext_d3d11va.h
=================================================================== ===================================================================
--- FFmpeg.orig/libavutil/hwcontext_d3d11va.h --- FFmpeg.orig/libavutil/hwcontext_d3d11va.h
+++ FFmpeg/libavutil/hwcontext_d3d11va.h +++ FFmpeg/libavutil/hwcontext_d3d11va.h
@@ -94,6 +94,16 @@ typedef struct AVD3D11VADeviceContext { @@ -108,6 +108,16 @@ typedef struct AVD3D11VADeviceContext {
void (*lock)(void *lock_ctx); * It applies globally to all AVD3D11VAFramesContext allocated from this device context.
void (*unlock)(void *lock_ctx); */
void *lock_ctx; UINT MiscFlags;
+ +
+ /** + /**
+ * DXGI adapter description of the device. + * DXGI adapter description of the device.
@@ -11,7 +11,7 @@ Index: FFmpeg/libavfilter/vf_overlay_cuda.c
AV_PIX_FMT_NONE, AV_PIX_FMT_NONE,
}; };
@@ -104,12 +106,14 @@ typedef struct OverlayCUDAContext { @@ -98,12 +100,14 @@ typedef struct OverlayCUDAContext {
enum AVPixelFormat in_format_overlay; enum AVPixelFormat in_format_overlay;
enum AVPixelFormat in_format_main; enum AVPixelFormat in_format_main;
@@ -28,7 +28,7 @@ Index: FFmpeg/libavfilter/vf_overlay_cuda.c
CUstream cu_stream; CUstream cu_stream;
FFFrameSync fs; FFFrameSync fs;
@@ -117,6 +121,7 @@ typedef struct OverlayCUDAContext { @@ -111,6 +115,7 @@ typedef struct OverlayCUDAContext {
int eval_mode; int eval_mode;
int x_position; int x_position;
int y_position; int y_position;
@@ -36,7 +36,7 @@ Index: FFmpeg/libavfilter/vf_overlay_cuda.c
double var_values[VAR_VARS_NB]; double var_values[VAR_VARS_NB];
char *x_expr, *y_expr; char *x_expr, *y_expr;
@@ -184,7 +189,10 @@ static int set_expr(AVExpr **pexpr, cons @@ -178,7 +183,10 @@ static int set_expr(AVExpr **pexpr, cons
static int formats_match(const enum AVPixelFormat format_main, const enum AVPixelFormat format_overlay) { static int formats_match(const enum AVPixelFormat format_main, const enum AVPixelFormat format_overlay) {
switch(format_main) { switch(format_main) {
case AV_PIX_FMT_NV12: case AV_PIX_FMT_NV12:
@@ -48,7 +48,7 @@ Index: FFmpeg/libavfilter/vf_overlay_cuda.c
case AV_PIX_FMT_YUV420P: case AV_PIX_FMT_YUV420P:
return format_overlay == AV_PIX_FMT_YUV420P || return format_overlay == AV_PIX_FMT_YUV420P ||
format_overlay == AV_PIX_FMT_YUVA420P; format_overlay == AV_PIX_FMT_YUVA420P;
@@ -199,26 +207,34 @@ static int formats_match(const enum AVPi @@ -193,26 +201,34 @@ static int formats_match(const enum AVPi
static int overlay_cuda_call_kernel( static int overlay_cuda_call_kernel(
OverlayCUDAContext *ctx, OverlayCUDAContext *ctx,
int x_position, int y_position, int x_position, int y_position,
@@ -88,7 +88,7 @@ Index: FFmpeg/libavfilter/vf_overlay_cuda.c
DIV_UP(main_width, BLOCK_X), DIV_UP(main_height, BLOCK_Y), 1, DIV_UP(main_width, BLOCK_X), DIV_UP(main_height, BLOCK_Y), 1,
BLOCK_X, BLOCK_Y, 1, BLOCK_X, BLOCK_Y, 1,
0, ctx->cu_stream, kernel_args, NULL)); 0, ctx->cu_stream, kernel_args, NULL));
@@ -242,8 +258,6 @@ static int overlay_cuda_blend(FFFrameSyn @@ -236,8 +252,6 @@ static int overlay_cuda_blend(FFFrameSyn
AVFrame *input_main, *input_overlay; AVFrame *input_main, *input_overlay;
@@ -97,7 +97,7 @@ Index: FFmpeg/libavfilter/vf_overlay_cuda.c
// read main and overlay frames from inputs // read main and overlay frames from inputs
ret = ff_framesync_dualinput_get(fs, &input_main, &input_overlay); ret = ff_framesync_dualinput_get(fs, &input_main, &input_overlay);
if (ret < 0) if (ret < 0)
@@ -300,11 +314,15 @@ FF_ENABLE_DEPRECATION_WARNINGS @@ -285,11 +299,15 @@ static int overlay_cuda_blend(FFFrameSyn
overlay_cuda_call_kernel(ctx, overlay_cuda_call_kernel(ctx,
ctx->x_position, ctx->y_position, ctx->x_position, ctx->y_position,
@@ -116,7 +116,7 @@ Index: FFmpeg/libavfilter/vf_overlay_cuda.c
// overlay rest planes depending on pixel format // overlay rest planes depending on pixel format
@@ -312,29 +330,46 @@ FF_ENABLE_DEPRECATION_WARNINGS @@ -297,29 +315,46 @@ static int overlay_cuda_blend(FFFrameSyn
case AV_PIX_FMT_NV12: case AV_PIX_FMT_NV12:
overlay_cuda_call_kernel(ctx, overlay_cuda_call_kernel(ctx,
ctx->x_position, ctx->y_position / 2, ctx->x_position, ctx->y_position / 2,
@@ -181,7 +181,7 @@ Index: FFmpeg/libavfilter/vf_overlay_cuda.c
break; break;
default: default:
av_log(ctx, AV_LOG_ERROR, "Passed unsupported overlay pixel format\n"); av_log(ctx, AV_LOG_ERROR, "Passed unsupported overlay pixel format\n");
@@ -406,7 +441,7 @@ static av_cold void overlay_cuda_uninit( @@ -388,7 +423,7 @@ static av_cold void overlay_cuda_uninit(
if (ctx->hwctx && ctx->cu_module) { if (ctx->hwctx && ctx->cu_module) {
CUcontext dummy; CUcontext dummy;
CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; CudaFunctions *cu = ctx->hwctx->internal->cuda_dl;
@@ -190,7 +190,7 @@ Index: FFmpeg/libavfilter/vf_overlay_cuda.c
CHECK_CU(cu->cuModuleUnload(ctx->cu_module)); CHECK_CU(cu->cuModuleUnload(ctx->cu_module));
CHECK_CU(cu->cuCtxPopCurrent(&dummy)); CHECK_CU(cu->cuCtxPopCurrent(&dummy));
} }
@@ -465,6 +500,8 @@ static int overlay_cuda_config_output(AV @@ -447,6 +482,8 @@ static int overlay_cuda_config_output(AV
return AVERROR(ENOSYS); return AVERROR(ENOSYS);
} }
@@ -199,7 +199,7 @@ Index: FFmpeg/libavfilter/vf_overlay_cuda.c
// check overlay input formats // check overlay input formats
if (!frames_ctx_overlay) { if (!frames_ctx_overlay) {
@@ -518,7 +555,13 @@ static int overlay_cuda_config_output(AV @@ -500,7 +537,13 @@ static int overlay_cuda_config_output(AV
return err; return err;
} }
@@ -214,7 +214,7 @@ Index: FFmpeg/libavfilter/vf_overlay_cuda.c
if (err < 0) { if (err < 0) {
CHECK_CU(cu->cuCtxPopCurrent(&dummy)); CHECK_CU(cu->cuCtxPopCurrent(&dummy));
return err; return err;
@@ -543,6 +586,9 @@ static int overlay_cuda_config_output(AV @@ -525,6 +568,9 @@ static int overlay_cuda_config_output(AV
static const AVOption overlay_cuda_options[] = { static const AVOption overlay_cuda_options[] = {
{ "x", "set the x expression of overlay", OFFSET(x_expr), AV_OPT_TYPE_STRING, { .str = "0" }, 0, 0, FLAGS }, { "x", "set the x expression of overlay", OFFSET(x_expr), AV_OPT_TYPE_STRING, { .str = "0" }, 0, 0, FLAGS },
{ "y", "set the y expression of overlay", OFFSET(y_expr), AV_OPT_TYPE_STRING, { .str = "0" }, 0, 0, FLAGS }, { "y", "set the y expression of overlay", OFFSET(y_expr), AV_OPT_TYPE_STRING, { .str = "0" }, 0, 0, FLAGS },
@@ -2,27 +2,21 @@ Index: FFmpeg/libavcodec/nvenc.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavcodec/nvenc.c --- FFmpeg.orig/libavcodec/nvenc.c
+++ FFmpeg/libavcodec/nvenc.c +++ FFmpeg/libavcodec/nvenc.c
@@ -24,6 +24,7 @@ @@ -24,6 +24,9 @@
#include "nvenc.h" #include "nvenc.h"
#include "hevc/sei.h" #include "hevc/sei.h"
+#ifndef NVENC_HAVE_HEVC_AND_AV1_MASTERING_METADATA
+#include "put_bits.h" +#include "put_bits.h"
+#endif
#if CONFIG_AV1_NVENC_ENCODER #if CONFIG_AV1_NVENC_ENCODER
#include "av1.h" #include "av1.h"
#endif #endif
@@ -32,6 +33,7 @@ @@ -2902,6 +2905,82 @@ static int prepare_sei_data_array(AVCode
#include "libavutil/hwcontext.h"
#include "libavutil/cuda_check.h"
#include "libavutil/imgutils.h"
+#include "libavutil/mastering_display_metadata.h"
#include "libavutil/mem.h"
#include "libavutil/pixdesc.h"
#include "libavutil/mathematics.h"
@@ -2639,6 +2641,80 @@ static int prepare_sei_data_array(AVCode
}
} }
} }
+
+#ifndef NVENC_HAVE_HEVC_AND_AV1_MASTERING_METADATA
+ if (avctx->codec->id == AV_CODEC_ID_HEVC) { + if (avctx->codec->id == AV_CODEC_ID_HEVC) {
+ AVFrameSideData *sd = av_frame_get_side_data(frame, AV_FRAME_DATA_MASTERING_DISPLAY_METADATA); + AVFrameSideData *sd = av_frame_get_side_data(frame, AV_FRAME_DATA_MASTERING_DISPLAY_METADATA);
+ +
@@ -96,6 +90,8 @@ Index: FFmpeg/libavcodec/nvenc.c
+ } + }
+ } + }
+ } + }
+#endif
+
if (!ctx->udu_sei) if (!ctx->udu_sei)
return sei_count; return sei_count;
@@ -11,7 +11,7 @@ Index: FFmpeg/libavcodec/d3d12va_hevc.c
DXVA_Qmatrix_HEVC qm; DXVA_Qmatrix_HEVC qm;
unsigned slice_count; unsigned slice_count;
DXVA_Slice_HEVC_Short slice_short[MAX_SLICES]; DXVA_Slice_HEVC_Short slice_short[MAX_SLICES];
@@ -151,12 +151,12 @@ static int d3d12va_hevc_end_frame(AVCode @@ -154,12 +154,12 @@ static int d3d12va_hevc_end_frame(AVCode
HEVCContext *h = avctx->priv_data; HEVCContext *h = avctx->priv_data;
HEVCDecodePictureContext *ctx_pic = h->cur_frame->hwaccel_picture_private; HEVCDecodePictureContext *ctx_pic = h->cur_frame->hwaccel_picture_private;
@@ -167,7 +167,7 @@ Index: FFmpeg/libavcodec/dxva2.c
} }
} }
@@ -627,8 +688,7 @@ int ff_dxva2_common_frame_params(AVCodec @@ -637,8 +698,7 @@ int ff_dxva2_common_frame_params(AVCodec
else else
num_surfaces += 2 + 4; /* 4 base work surface in vpp async */ num_surfaces += 2 + 4; /* 4 base work surface in vpp async */
@@ -243,7 +243,7 @@ Index: FFmpeg/libavcodec/dxva2_hevc.c
pp->wFormatAndSequenceInfoFlags = (sps->chroma_format_idc << 0) | pp->wFormatAndSequenceInfoFlags = (sps->chroma_format_idc << 0) |
(sps->separate_colour_plane << 2) | (sps->separate_colour_plane << 2) |
((sps->bit_depth - 8) << 3) | ((sps->bit_depth - 8) << 3) |
@@ -413,14 +443,15 @@ static int dxva2_hevc_end_frame(AVCodecC @@ -414,14 +444,15 @@ static int dxva2_hevc_end_frame(AVCodecC
{ {
HEVCContext *h = avctx->priv_data; HEVCContext *h = avctx->priv_data;
struct hevc_dxva2_picture_context *ctx_pic = h->cur_frame->hwaccel_picture_private; struct hevc_dxva2_picture_context *ctx_pic = h->cur_frame->hwaccel_picture_private;
@@ -324,22 +324,20 @@ Index: FFmpeg/libavcodec/dxva2_internal.h
#if CONFIG_D3D12VA #if CONFIG_D3D12VA
unsigned ff_d3d12va_get_surface_index(const AVCodecContext *avctx, unsigned ff_d3d12va_get_surface_index(const AVCodecContext *avctx,
D3D12VADecodeContext *ctx, const AVFrame *frame, D3D12VADecodeContext *ctx, const AVFrame *frame,
@@ -171,7 +223,9 @@ void ff_dxva2_h264_fill_picture_paramete @@ -172,7 +224,7 @@ void ff_dxva2_h264_fill_picture_paramete
void ff_dxva2_h264_fill_scaling_lists(const AVCodecContext *avctx, AVDXVAContext *ctx, DXVA_Qmatrix_H264 *qm); void ff_dxva2_h264_fill_scaling_lists(const AVCodecContext *avctx, AVDXVAContext *ctx, DXVA_Qmatrix_H264 *qm);
#if CONFIG_HEVC_D3D12VA_HWACCEL || CONFIG_HEVC_D3D11VA_HWACCEL || CONFIG_HEVC_D3D11VA2_HWACCEL || CONFIG_HEVC_DXVA2_HWACCEL
-void ff_dxva2_hevc_fill_picture_parameters(const AVCodecContext *avctx, AVDXVAContext *ctx, DXVA_PicParams_HEVC *pp); -void ff_dxva2_hevc_fill_picture_parameters(const AVCodecContext *avctx, AVDXVAContext *ctx, DXVA_PicParams_HEVC *pp);
+#if CONFIG_HEVC_D3D12VA_HWACCEL || CONFIG_HEVC_D3D11VA_HWACCEL || CONFIG_HEVC_D3D11VA2_HWACCEL || CONFIG_HEVC_DXVA2_HWACCEL
+void ff_dxva2_hevc_fill_picture_parameters(const AVCodecContext *avctx, AVDXVAContext *ctx, ff_DXVA_PicParams_HEVC_Rext *ppext); +void ff_dxva2_hevc_fill_picture_parameters(const AVCodecContext *avctx, AVDXVAContext *ctx, ff_DXVA_PicParams_HEVC_Rext *ppext);
+#endif
void ff_dxva2_hevc_fill_scaling_lists(const AVCodecContext *avctx, AVDXVAContext *ctx, DXVA_Qmatrix_HEVC *qm); void ff_dxva2_hevc_fill_scaling_lists(const AVCodecContext *avctx, AVDXVAContext *ctx, DXVA_Qmatrix_HEVC *qm);
#endif
Index: FFmpeg/libavcodec/hevc/hevcdec.c Index: FFmpeg/libavcodec/hevc/hevcdec.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavcodec/hevc/hevcdec.c --- FFmpeg.orig/libavcodec/hevc/hevcdec.c
+++ FFmpeg/libavcodec/hevc/hevcdec.c +++ FFmpeg/libavcodec/hevc/hevcdec.c
@@ -592,6 +592,13 @@ static enum AVPixelFormat get_format(HEV @@ -648,6 +648,13 @@ static enum AVPixelFormat get_format(HEV
#endif #endif
break; break;
case AV_PIX_FMT_YUV444P: case AV_PIX_FMT_YUV444P:
@@ -353,7 +351,7 @@ Index: FFmpeg/libavcodec/hevc/hevcdec.c
#if CONFIG_HEVC_VAAPI_HWACCEL #if CONFIG_HEVC_VAAPI_HWACCEL
*fmt++ = AV_PIX_FMT_VAAPI; *fmt++ = AV_PIX_FMT_VAAPI;
#endif #endif
@@ -610,6 +617,13 @@ static enum AVPixelFormat get_format(HEV @@ -666,6 +673,13 @@ static enum AVPixelFormat get_format(HEV
break; break;
case AV_PIX_FMT_YUV422P: case AV_PIX_FMT_YUV422P:
case AV_PIX_FMT_YUV422P10LE: case AV_PIX_FMT_YUV422P10LE:
@@ -367,7 +365,7 @@ Index: FFmpeg/libavcodec/hevc/hevcdec.c
#if CONFIG_HEVC_VAAPI_HWACCEL #if CONFIG_HEVC_VAAPI_HWACCEL
*fmt++ = AV_PIX_FMT_VAAPI; *fmt++ = AV_PIX_FMT_VAAPI;
#endif #endif
@@ -627,6 +641,13 @@ static enum AVPixelFormat get_format(HEV @@ -686,6 +700,13 @@ static enum AVPixelFormat get_format(HEV
/* NOTE: fallthrough */ /* NOTE: fallthrough */
case AV_PIX_FMT_YUV420P12: case AV_PIX_FMT_YUV420P12:
case AV_PIX_FMT_YUV444P12: case AV_PIX_FMT_YUV444P12:
@@ -381,7 +379,7 @@ Index: FFmpeg/libavcodec/hevc/hevcdec.c
#if CONFIG_HEVC_VAAPI_HWACCEL #if CONFIG_HEVC_VAAPI_HWACCEL
*fmt++ = AV_PIX_FMT_VAAPI; *fmt++ = AV_PIX_FMT_VAAPI;
#endif #endif
@@ -641,6 +662,13 @@ static enum AVPixelFormat get_format(HEV @@ -700,6 +721,13 @@ static enum AVPixelFormat get_format(HEV
#endif #endif
break; break;
case AV_PIX_FMT_YUV422P12: case AV_PIX_FMT_YUV422P12:
@@ -0,0 +1,121 @@
Index: FFmpeg/libavfilter/vf_scale_vt.c
===================================================================
--- FFmpeg.orig/libavfilter/vf_scale_vt.c
+++ FFmpeg/libavfilter/vf_scale_vt.c
@@ -41,11 +41,26 @@ typedef struct ScaleVtContext {
enum AVColorPrimaries colour_primaries;
enum AVColorTransferCharacteristic colour_transfer;
enum AVColorSpace colour_matrix;
+ enum AVPixelFormat format;
char *colour_primaries_string;
char *colour_transfer_string;
char *colour_matrix_string;
} ScaleVtContext;
+static const enum AVPixelFormat supported_formats[] = {
+ AV_PIX_FMT_NV12,
+ AV_PIX_FMT_P010,
+ AV_PIX_FMT_NONE,
+};
+
+static int format_is_supported(enum AVPixelFormat fmt)
+{
+ for (int i = 0; i < FF_ARRAY_ELEMS(supported_formats); i++)
+ if (supported_formats[i] == fmt)
+ return 1;
+ return 0;
+}
+
static av_cold int scale_vt_init(AVFilterContext *avctx)
{
ScaleVtContext *s = avctx->priv;
@@ -58,24 +73,6 @@ static av_cold int scale_vt_init(AVFilte
return AVERROR_EXTERNAL;
}
-#define STRING_OPTION(var_name, func_name, default_value) \
- do { \
- if (s->var_name##_string) { \
- int var = av_##func_name##_from_name(s->var_name##_string); \
- if (var < 0) { \
- av_log(avctx, AV_LOG_ERROR, "Invalid %s.\n", #var_name); \
- return AVERROR(EINVAL); \
- } \
- s->var_name = var; \
- } else { \
- s->var_name = default_value; \
- } \
- } while (0)
-
- STRING_OPTION(colour_primaries, color_primaries, AVCOL_PRI_UNSPECIFIED);
- STRING_OPTION(colour_transfer, color_transfer, AVCOL_TRC_UNSPECIFIED);
- STRING_OPTION(colour_matrix, color_space, AVCOL_SPC_UNSPECIFIED);
-
if (s->colour_primaries != AVCOL_PRI_UNSPECIFIED) {
value = av_map_videotoolbox_color_primaries_from_av(s->colour_primaries);
if (!value) {
@@ -241,6 +238,35 @@ static int scale_vt_config_output(AVFilt
FilterLink *inl = ff_filter_link(inlink);
AVHWFramesContext *hw_frame_ctx_in;
AVHWFramesContext *hw_frame_ctx_out;
+ enum AVPixelFormat out_format;
+
+#define STRING_OPTION(var_name, func_name, default_value) \
+ do { \
+ if (s->var_name##_string) { \
+ int var = av_##func_name##_from_name(s->var_name##_string); \
+ if (var < 0) { \
+ av_log(avctx, AV_LOG_ERROR, "Invalid %s.\n", #var_name); \
+ return AVERROR(EINVAL); \
+ } \
+ s->var_name = var; \
+ } else { \
+ s->var_name = default_value; \
+ } \
+ } while (0)
+
+ STRING_OPTION(colour_primaries, color_primaries, AVCOL_PRI_UNSPECIFIED);
+ STRING_OPTION(colour_transfer, color_transfer, AVCOL_TRC_UNSPECIFIED);
+ STRING_OPTION(colour_matrix, color_space, AVCOL_SPC_UNSPECIFIED);
+
+ if (!((s->colour_primaries == AVCOL_PRI_UNSPECIFIED &&
+ s->colour_transfer == AVCOL_TRC_UNSPECIFIED) ||
+ (s->colour_primaries == AVCOL_PRI_BT2020 &&
+ s->colour_transfer == AVCOL_TRC_SMPTE2084))) {
+ av_frame_side_data_remove(&outlink->side_data, &outlink->nb_side_data,
+ AV_FRAME_DATA_MASTERING_DISPLAY_METADATA);
+ av_frame_side_data_remove(&outlink->side_data, &outlink->nb_side_data,
+ AV_FRAME_DATA_CONTENT_LIGHT_LEVEL);
+ }
err = ff_scale_eval_dimensions(s, s->w_expr, s->h_expr, inlink, outlink,
&s->output_width,
@@ -263,11 +289,18 @@ static int scale_vt_config_output(AVFilt
hw_frame_ctx_in = (AVHWFramesContext *)inl->hw_frames_ctx->data;
+ out_format = (s->format == AV_PIX_FMT_NONE) ? hw_frame_ctx_in->sw_format : s->format;
+ if (!format_is_supported(s->format)) {
+ av_log(s, AV_LOG_ERROR, "Unsupported output format: %s\n",
+ av_get_pix_fmt_name(out_format));
+ return AVERROR(ENOSYS);
+ }
+
av_buffer_unref(&outl->hw_frames_ctx);
outl->hw_frames_ctx = av_hwframe_ctx_alloc(hw_frame_ctx_in->device_ref);
hw_frame_ctx_out = (AVHWFramesContext *)outl->hw_frames_ctx->data;
hw_frame_ctx_out->format = AV_PIX_FMT_VIDEOTOOLBOX;
- hw_frame_ctx_out->sw_format = hw_frame_ctx_in->sw_format;
+ hw_frame_ctx_out->sw_format = out_format;
hw_frame_ctx_out->width = outlink->w;
hw_frame_ctx_out->height = outlink->h;
((AVVTFramesContext *)hw_frame_ctx_out->hwctx)->color_range = ((AVVTFramesContext *)hw_frame_ctx_in->hwctx)->color_range;
@@ -300,6 +333,8 @@ static const AVOption scale_vt_options[]
OFFSET(colour_primaries_string), AV_OPT_TYPE_STRING, { .str = NULL }, .flags = FLAGS },
{ "color_transfer", "Output colour transfer characteristics",
OFFSET(colour_transfer_string), AV_OPT_TYPE_STRING, { .str = NULL }, .flags = FLAGS },
+ { "format", "Output pixel format",
+ OFFSET(format), AV_OPT_TYPE_PIXEL_FMT, { .i64 = AV_PIX_FMT_NONE }, AV_PIX_FMT_NONE, INT_MAX, .flags = FLAGS },
{ NULL },
};
@@ -2,118 +2,55 @@ Index: FFmpeg/configure
=================================================================== ===================================================================
--- FFmpeg.orig/configure --- FFmpeg.orig/configure
+++ FFmpeg/configure +++ FFmpeg/configure
@@ -3928,6 +3928,7 @@ overlay_opencl_filter_deps="opencl" @@ -4191,6 +4191,7 @@ overlay_qsv_filter_deps="libmfx"
overlay_qsv_filter_deps="libmfx"
overlay_qsv_filter_select="qsvvpp" overlay_qsv_filter_select="qsvvpp"
overlay_rkrga_filter_deps="rkrga"
overlay_vaapi_filter_deps="vaapi VAProcPipelineCaps_blend_flags" overlay_vaapi_filter_deps="vaapi VAProcPipelineCaps_blend_flags"
+overlay_videotoolbox_filter_deps="metal corevideo coreimage videotoolbox" +overlay_videotoolbox_filter_deps="metal corevideo coreimage videotoolbox"
overlay_vulkan_filter_deps="vulkan spirv_compiler" overlay_vulkan_filter_deps="vulkan spirv_library"
overlay_rkrga_filter_deps="rkrga"
owdenoise_filter_deps="gpl" owdenoise_filter_deps="gpl"
Index: FFmpeg/doc/filters.texi pad_opencl_filter_deps="opencl"
===================================================================
--- FFmpeg.orig/doc/filters.texi
+++ FFmpeg/doc/filters.texi
@@ -19050,6 +19050,58 @@ See @ref{framesync}.
This filter also supports the @ref{framesync} options.
+@section overlay_videotoolbox
+
+Overlay one video on top of another.
+
+This is the VideoToolbox variant of the @ref{overlay} filter.
+It takes two inputs and has one output. The first input is the "main" video on which the second input is overlaid.
+It only accepts VideoToolbox frames. The underlying input pixel formats do not have to match.
+Different input pixel formats and color spaces will be automatically converted using hardware accelerated methods.
+The final output will have the same pixel format and color space as the "main" input.
+
+The filter accepts the following options:
+
+@table @option
+
+@item x
+Set the x coordinate of the overlaid video on the main video.
+Default value is @code{0}.
+
+@item y
+Set the y coordinate of the overlaid video on the main video.
+Default value is @code{0}.
+
+@item eof_action
+See @ref{framesync}.
+
+@item shortest
+See @ref{framesync}.
+
+@item repeatlast
+See @ref{framesync}.
+
+@end table
+
+@subsection Examples
+
+@itemize
+@item
+Overlay an image LOGO at the top-left corner of the INPUT video.
+The INPUT video is in nv12 format and the LOGO image is in rgba format.
+@example
+-hwaccel videotoolbox -i INPUT -i LOGO -codec:v:0 h264_videotoolbox -filter_complex "[0:v]format=nv12,hwupload[a], [1:v]format=rgba,hwupload[b], [a][b]overlay_videotoolbox" OUTPUT
+@end example
+@item
+Overlay an SDR video OVERLAY at the top-left corner of the HDR video MAIN.
+The INPUT video is in p010 format and the LOGO image is in nv12 format.
+The OUTPUT video will also be an HDR video with OVERLAY mapped to HDR.
+@example
+-hwaccel videotoolbox -i MAIN -i OVERLAY -codec:v:0 hevc_videotoolbox -tag:v hvc1 -filter_complex "[0:v]format=p010,hwupload[a], [1:v]format=nv12,hwupload[b], [a][b]overlay_videotoolbox" OUTPUT
+@end example
+
+@end itemize
+
@section owdenoise
Apply Overcomplete Wavelet denoiser.
Index: FFmpeg/libavfilter/Makefile Index: FFmpeg/libavfilter/Makefile
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/Makefile --- FFmpeg.orig/libavfilter/Makefile
+++ FFmpeg/libavfilter/Makefile +++ FFmpeg/libavfilter/Makefile
@@ -416,6 +416,9 @@ OBJS-$(CONFIG_OVERLAY_OPENCL_FILTER) @@ -429,6 +429,9 @@ OBJS-$(CONFIG_OVERLAY_OPENCL_FILTER)
opencl/overlay.o framesync.o
OBJS-$(CONFIG_OVERLAY_QSV_FILTER) += vf_overlay_qsv.o framesync.o OBJS-$(CONFIG_OVERLAY_QSV_FILTER) += vf_overlay_qsv.o framesync.o
OBJS-$(CONFIG_OVERLAY_RKRGA_FILTER) += vf_overlay_rkrga.o framesync.o
OBJS-$(CONFIG_OVERLAY_VAAPI_FILTER) += vf_overlay_vaapi.o framesync.o vaapi_vpp.o OBJS-$(CONFIG_OVERLAY_VAAPI_FILTER) += vf_overlay_vaapi.o framesync.o vaapi_vpp.o
+OBJS-$(CONFIG_OVERLAY_VIDEOTOOLBOX_FILTER) += vf_overlay_videotoolbox.o framesync.o \ +OBJS-$(CONFIG_OVERLAY_VIDEOTOOLBOX_FILTER) += vf_overlay_videotoolbox.o framesync.o \
+ metal/vf_overlay_videotoolbox.metallib.o \ + metal/vf_overlay_videotoolbox.metallib.o \
+ metal/utils.o + metal/utils.o
OBJS-$(CONFIG_OVERLAY_VULKAN_FILTER) += vf_overlay_vulkan.o vulkan.o vulkan_filter.o OBJS-$(CONFIG_OVERLAY_VULKAN_FILTER) += vf_overlay_vulkan.o vulkan.o vulkan_filter.o
OBJS-$(CONFIG_OVERLAY_RKRGA_FILTER) += vf_overlay_rkrga.o framesync.o
OBJS-$(CONFIG_OWDENOISE_FILTER) += vf_owdenoise.o OBJS-$(CONFIG_OWDENOISE_FILTER) += vf_owdenoise.o
OBJS-$(CONFIG_PAD_FILTER) += vf_pad.o
Index: FFmpeg/libavfilter/allfilters.c Index: FFmpeg/libavfilter/allfilters.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/allfilters.c --- FFmpeg.orig/libavfilter/allfilters.c
+++ FFmpeg/libavfilter/allfilters.c +++ FFmpeg/libavfilter/allfilters.c
@@ -389,6 +389,7 @@ extern const AVFilter ff_vf_overlay; @@ -399,6 +399,7 @@ extern const FFFilter ff_vf_overlay_open
extern const AVFilter ff_vf_overlay_opencl; extern const FFFilter ff_vf_overlay_qsv;
extern const AVFilter ff_vf_overlay_qsv; extern const FFFilter ff_vf_overlay_rkrga;
extern const AVFilter ff_vf_overlay_vaapi; extern const FFFilter ff_vf_overlay_vaapi;
+extern const AVFilter ff_vf_overlay_videotoolbox; +extern const FFFilter ff_vf_overlay_videotoolbox;
extern const AVFilter ff_vf_overlay_vulkan; extern const FFFilter ff_vf_overlay_vulkan;
extern const AVFilter ff_vf_overlay_cuda; extern const FFFilter ff_vf_overlay_cuda;
extern const AVFilter ff_vf_overlay_rkrga; extern const FFFilter ff_vf_owdenoise;
Index: FFmpeg/libavfilter/metal/utils.m Index: FFmpeg/libavfilter/metal/utils.m
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/metal/utils.m --- FFmpeg.orig/libavfilter/metal/utils.m
+++ FFmpeg/libavfilter/metal/utils.m +++ FFmpeg/libavfilter/metal/utils.m
@@ -55,6 +55,9 @@ CVMetalTextureRef ff_metal_texture_from_ @@ -56,6 +56,9 @@ CVMetalTextureRef ff_metal_texture_from_
{ {
CVMetalTextureRef tex = NULL; CVMetalTextureRef tex = NULL;
CVReturn ret; CVReturn ret;
+ bool is_planer = CVPixelBufferIsPlanar(pixbuf); + bool is_planar = CVPixelBufferIsPlanar(pixbuf);
+ size_t width = is_planer ? CVPixelBufferGetWidthOfPlane(pixbuf, plane) : CVPixelBufferGetWidth(pixbuf); + size_t width = is_planar ? CVPixelBufferGetWidthOfPlane(pixbuf, plane) : CVPixelBufferGetWidth(pixbuf);
+ size_t height = is_planer ? CVPixelBufferGetHeightOfPlane(pixbuf, plane) : CVPixelBufferGetHeight(pixbuf); + size_t height = is_planar ? CVPixelBufferGetHeightOfPlane(pixbuf, plane) : CVPixelBufferGetHeight(pixbuf);
ret = CVMetalTextureCacheCreateTextureFromImage( ret = CVMetalTextureCacheCreateTextureFromImage(
NULL, NULL,
@@ -62,8 +65,8 @@ CVMetalTextureRef ff_metal_texture_from_ @@ -63,8 +66,8 @@ CVMetalTextureRef ff_metal_texture_from_
pixbuf, pixbuf,
NULL, NULL,
format, format,
@@ -232,7 +169,7 @@ Index: FFmpeg/libavfilter/vf_overlay_videotoolbox.m
=================================================================== ===================================================================
--- /dev/null --- /dev/null
+++ FFmpeg/libavfilter/vf_overlay_videotoolbox.m +++ FFmpeg/libavfilter/vf_overlay_videotoolbox.m
@@ -0,0 +1,758 @@ @@ -0,0 +1,759 @@
+/* +/*
+ * Copyright (C) 2024 Gnattu OC <gnattuoc@me.com> + * Copyright (C) 2024 Gnattu OC <gnattuoc@me.com>
+ * + *
@@ -977,11 +914,11 @@ Index: FFmpeg/libavfilter/vf_overlay_videotoolbox.m
+ }, + },
+}; +};
+ +
+const AVFilter ff_vf_overlay_videotoolbox = { +const FFFilter ff_vf_overlay_videotoolbox = {
+ .name = "overlay_videotoolbox", + .p.name = "overlay_videotoolbox",
+ .description = NULL_IF_CONFIG_SMALL("Overlay filter for VideoToolbox frames using Metal compute"), + .p.description = NULL_IF_CONFIG_SMALL("Overlay filter for VideoToolbox frames using Metal compute"),
+ .priv_size = OVERLAY_VT_CTX_SIZE, + .priv_size = OVERLAY_VT_CTX_SIZE,
+ .priv_class = &overlay_videotoolbox_class, + .p.priv_class = &overlay_videotoolbox_class,
+ .init = overlay_videotoolbox_init, + .init = overlay_videotoolbox_init,
+ .uninit = overlay_videotoolbox_uninit, + .uninit = overlay_videotoolbox_uninit,
+ .activate = overlay_videotoolbox_activate, + .activate = overlay_videotoolbox_activate,
@@ -990,4 +927,5 @@ Index: FFmpeg/libavfilter/vf_overlay_videotoolbox.m
+ FILTER_INPUTS(overlay_videotoolbox_inputs), + FILTER_INPUTS(overlay_videotoolbox_inputs),
+ FILTER_OUTPUTS(overlay_videotoolbox_outputs), + FILTER_OUTPUTS(overlay_videotoolbox_outputs),
+ .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE, + .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE,
+ .p.flags = AVFILTER_FLAG_HWDEVICE,
+}; +};
@@ -2,13 +2,13 @@ Index: FFmpeg/configure
=================================================================== ===================================================================
--- FFmpeg.orig/configure --- FFmpeg.orig/configure
+++ FFmpeg/configure +++ FFmpeg/configure
@@ -3979,7 +3979,7 @@ tonemap_vaapi_filter_deps="vaapi VAProcF @@ -4243,7 +4243,7 @@ tonemap_vaapi_filter_deps="vaapi VAProcF
tonemap_opencl_filter_deps="opencl const_nan" tonemap_opencl_filter_deps="opencl const_nan"
transpose_opencl_filter_deps="opencl" transpose_opencl_filter_deps="opencl"
transpose_vaapi_filter_deps="vaapi VAProcPipelineCaps_rotation_flags" transpose_vaapi_filter_deps="vaapi VAProcPipelineCaps_rotation_flags"
-transpose_vt_filter_deps="videotoolbox VTPixelRotationSessionCreate" -transpose_vt_filter_deps="videotoolbox VTPixelRotationSessionCreate"
+transpose_vt_filter_deps="coreimage videotoolbox" +transpose_vt_filter_deps="coreimage videotoolbox"
transpose_vulkan_filter_deps="vulkan spirv_compiler" transpose_vulkan_filter_deps="vulkan spirv_library"
unsharp_opencl_filter_deps="opencl" unsharp_opencl_filter_deps="opencl"
uspp_filter_deps="gpl avcodec" uspp_filter_deps="gpl avcodec"
Index: FFmpeg/libavfilter/vf_transpose_vt.c Index: FFmpeg/libavfilter/vf_transpose_vt.c
@@ -294,24 +294,24 @@ Index: FFmpeg/libavfilter/vf_transpose_vt.c
- }, - },
-}; -};
- -
-const AVFilter ff_vf_transpose_vt = { -const FFFilter ff_vf_transpose_vt = {
- .name = "transpose_vt", - .p.name = "transpose_vt",
- .description = NULL_IF_CONFIG_SMALL("Transpose Videotoolbox frames"), - .p.description = NULL_IF_CONFIG_SMALL("Transpose Videotoolbox frames"),
- .p.priv_class = &transpose_vt_class,
- .p.flags = AVFILTER_FLAG_HWDEVICE,
- .priv_size = sizeof(TransposeVtContext), - .priv_size = sizeof(TransposeVtContext),
- .init = transpose_vt_init, - .init = transpose_vt_init,
- .uninit = transpose_vt_uninit, - .uninit = transpose_vt_uninit,
- FILTER_INPUTS(transpose_vt_inputs), - FILTER_INPUTS(transpose_vt_inputs),
- FILTER_OUTPUTS(transpose_vt_outputs), - FILTER_OUTPUTS(transpose_vt_outputs),
- FILTER_SINGLE_PIXFMT(AV_PIX_FMT_VIDEOTOOLBOX), - FILTER_SINGLE_PIXFMT(AV_PIX_FMT_VIDEOTOOLBOX),
- .priv_class = &transpose_vt_class,
- .flags = AVFILTER_FLAG_HWDEVICE,
- .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE, - .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE,
-}; -};
Index: FFmpeg/libavfilter/vf_transpose_vt.m Index: FFmpeg/libavfilter/vf_transpose_vt.m
=================================================================== ===================================================================
--- /dev/null --- /dev/null
+++ FFmpeg/libavfilter/vf_transpose_vt.m +++ FFmpeg/libavfilter/vf_transpose_vt.m
@@ -0,0 +1,383 @@ @@ -0,0 +1,384 @@
+/* +/*
+ * Copyright (c) 2023 Zhao Zhili <zhilizhao@tencent.com> + * Copyright (c) 2023 Zhao Zhili <zhilizhao@tencent.com>
+ * Copyright (c) 2024 Gnattu OC <gnattuoc@me.com> + * Copyright (c) 2024 Gnattu OC <gnattuoc@me.com>
@@ -683,15 +683,16 @@ Index: FFmpeg/libavfilter/vf_transpose_vt.m
+ }, + },
+}; +};
+ +
+const AVFilter ff_vf_transpose_vt = { +const FFFilter ff_vf_transpose_vt = {
+ .name = "transpose_vt", + .p.name = "transpose_vt",
+ .description = NULL_IF_CONFIG_SMALL("Transpose Videotoolbox frames"), + .p.description = NULL_IF_CONFIG_SMALL("Transpose Videotoolbox frames"),
+ .priv_size = sizeof(TransposeVtContext), + .priv_size = sizeof(TransposeVtContext),
+ .init = transpose_vt_init, + .init = transpose_vt_init,
+ .uninit = transpose_vt_uninit, + .uninit = transpose_vt_uninit,
+ FILTER_INPUTS(transpose_vt_inputs), + FILTER_INPUTS(transpose_vt_inputs),
+ FILTER_OUTPUTS(transpose_vt_outputs), + FILTER_OUTPUTS(transpose_vt_outputs),
+ FILTER_SINGLE_PIXFMT(AV_PIX_FMT_VIDEOTOOLBOX), + FILTER_SINGLE_PIXFMT(AV_PIX_FMT_VIDEOTOOLBOX),
+ .priv_class = &transpose_vt_class, + .p.priv_class = &transpose_vt_class,
+ .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE, + .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE,
+ .p.flags = AVFILTER_FLAG_HWDEVICE,
+}; +};
@@ -1,13 +0,0 @@
Index: FFmpeg/libavformat/movenc.c
===================================================================
--- FFmpeg.orig/libavformat/movenc.c
+++ FFmpeg/libavformat/movenc.c
@@ -3202,7 +3202,7 @@ static int mov_write_stbl_tag(AVFormatCo
track->par->codec_tag == MKTAG('r','t','p',' ')) &&
track->has_keyframes && track->has_keyframes < track->entry)
mov_write_stss_tag(pb, track, MOV_SYNC_SAMPLE);
- if (track->par->codec_type == AVMEDIA_TYPE_VIDEO && track->has_disposable)
+ if (track->par->codec_type == AVMEDIA_TYPE_VIDEO && track->has_disposable && track->entry)
mov_write_sdtp_tag(pb, track);
if (track->mode == MODE_MOV && track->flags & MOV_TRACK_STPS)
mov_write_stss_tag(pb, track, MOV_PARTIAL_SYNC_SAMPLE);
@@ -2,7 +2,7 @@ Index: FFmpeg/configure
=================================================================== ===================================================================
--- FFmpeg.orig/configure --- FFmpeg.orig/configure
+++ FFmpeg/configure +++ FFmpeg/configure
@@ -3976,6 +3976,7 @@ tinterlace_merge_test_deps="tinterlace_f @@ -4240,6 +4240,7 @@ tinterlace_merge_test_deps="tinterlace_f
tinterlace_pad_test_deps="tinterlace_filter" tinterlace_pad_test_deps="tinterlace_filter"
tonemap_filter_deps="const_nan" tonemap_filter_deps="const_nan"
tonemap_vaapi_filter_deps="vaapi VAProcFilterParameterBufferHDRToneMapping" tonemap_vaapi_filter_deps="vaapi VAProcFilterParameterBufferHDRToneMapping"
@@ -14,7 +14,7 @@ Index: FFmpeg/libavfilter/Makefile
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/Makefile --- FFmpeg.orig/libavfilter/Makefile
+++ FFmpeg/libavfilter/Makefile +++ FFmpeg/libavfilter/Makefile
@@ -537,6 +537,9 @@ OBJS-$(CONFIG_TONEMAP_CUDA_FILTER) @@ -554,6 +554,9 @@ OBJS-$(CONFIG_TONEMAP_CUDA_FILTER)
OBJS-$(CONFIG_TONEMAP_OPENCL_FILTER) += vf_tonemap_opencl.o opencl.o \ OBJS-$(CONFIG_TONEMAP_OPENCL_FILTER) += vf_tonemap_opencl.o opencl.o \
opencl/tonemap.o opencl/colorspace_common.o opencl/tonemap.o opencl/colorspace_common.o
OBJS-$(CONFIG_TONEMAP_VAAPI_FILTER) += vf_tonemap_vaapi.o vaapi_vpp.o OBJS-$(CONFIG_TONEMAP_VAAPI_FILTER) += vf_tonemap_vaapi.o vaapi_vpp.o
@@ -28,14 +28,14 @@ Index: FFmpeg/libavfilter/allfilters.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/allfilters.c --- FFmpeg.orig/libavfilter/allfilters.c
+++ FFmpeg/libavfilter/allfilters.c +++ FFmpeg/libavfilter/allfilters.c
@@ -502,6 +502,7 @@ extern const AVFilter ff_vf_tonemap; @@ -517,6 +517,7 @@ extern const FFFilter ff_vf_tonemap;
extern const AVFilter ff_vf_tonemap_cuda; extern const FFFilter ff_vf_tonemap_cuda;
extern const AVFilter ff_vf_tonemap_opencl; extern const FFFilter ff_vf_tonemap_opencl;
extern const AVFilter ff_vf_tonemap_vaapi; extern const FFFilter ff_vf_tonemap_vaapi;
+extern const AVFilter ff_vf_tonemap_videotoolbox; +extern const FFFilter ff_vf_tonemap_videotoolbox;
extern const AVFilter ff_vf_tpad; extern const FFFilter ff_vf_tpad;
extern const AVFilter ff_vf_transpose; extern const FFFilter ff_vf_transpose;
extern const AVFilter ff_vf_transpose_npp; extern const FFFilter ff_vf_transpose_npp;
Index: FFmpeg/libavfilter/metal/vf_tonemap_videotoolbox.metal Index: FFmpeg/libavfilter/metal/vf_tonemap_videotoolbox.metal
=================================================================== ===================================================================
--- /dev/null --- /dev/null
@@ -966,7 +966,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_videotoolbox.m
=================================================================== ===================================================================
--- /dev/null --- /dev/null
+++ FFmpeg/libavfilter/vf_tonemap_videotoolbox.m +++ FFmpeg/libavfilter/vf_tonemap_videotoolbox.m
@@ -0,0 +1,1211 @@ @@ -0,0 +1,1221 @@
+/* +/*
+ * Copyright (c) 2024 Gnattu OC <gnattuoc@me.com> + * Copyright (c) 2024 Gnattu OC <gnattuoc@me.com>
+ * + *
@@ -1061,9 +1061,11 @@ Index: FFmpeg/libavfilter/vf_tonemap_videotoolbox.m
+ enum AVPixelFormat in_fmt, out_fmt; + enum AVPixelFormat in_fmt, out_fmt;
+ const AVPixFmtDescriptor *in_desc, *out_desc; + const AVPixFmtDescriptor *in_desc, *out_desc;
+ int in_planes, out_planes; + int in_planes, out_planes;
+ struct DoviMetadata *dovi; + struct FFDOVIMetadataRemap *dovi;
+ enum TonemapAlgorithm tonemap; + /* enum TonemapAlgorithm */
+ enum TonemapMode tonemap_mode; + int tonemap;
+ /* enum TonemapMode */
+ int tonemap_mode;
+ enum AVPixelFormat format; + enum AVPixelFormat format;
+ int apply_dovi; + int apply_dovi;
+ double peak; + double peak;
@@ -1087,12 +1089,12 @@ Index: FFmpeg/libavfilter/vf_tonemap_videotoolbox.m
+ CVMetalTextureCacheRef texture_cache; + CVMetalTextureCacheRef texture_cache;
+} TonemapVideoToolboxContext; +} TonemapVideoToolboxContext;
+ +
+static const short linearize_funcs[AVCOL_TRC_NB] = { +static const short linearize_funcs[] = {
+ [AVCOL_TRC_SMPTE2084] = 1, //"eotf_st2084", + [AVCOL_TRC_SMPTE2084] = 1, //"eotf_st2084",
+ [AVCOL_TRC_ARIB_STD_B67] = 2, //"eotf_arib_b67", + [AVCOL_TRC_ARIB_STD_B67] = 2, //"eotf_arib_b67",
+}; +};
+ +
+static const short delinearize_funcs[AVCOL_TRC_NB] = { +static const short delinearize_funcs[] = {
+ [AVCOL_TRC_BT709] = 1, //"inverse_eotf_bt1886", + [AVCOL_TRC_BT709] = 1, //"inverse_eotf_bt1886",
+ [AVCOL_TRC_BT2020_10] = 1, //"inverse_eotf_bt1886", + [AVCOL_TRC_BT2020_10] = 1, //"inverse_eotf_bt1886",
+}; +};
@@ -1171,7 +1173,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_videotoolbox.m
+ for (c = 0; c < 3; c++) { + for (c = 0; c < 3; c++) {
+ int has_poly = 0, has_mmr = 0, mmr_single = 1; + int has_poly = 0, has_mmr = 0, mmr_single = 1;
+ int mmr_idx = 0, min_order = 3, max_order = 1; + int mmr_idx = 0, min_order = 3, max_order = 1;
+ const struct ReshapeData *comp = &ctx->dovi->comp[c]; + const struct FFDOVIReshapeData *comp = &ctx->dovi->comp[c];
+ if (!comp->num_pivots) + if (!comp->num_pivots)
+ continue; + continue;
+ av_assert0(comp->num_pivots >= 2 && comp->num_pivots <= 9); + av_assert0(comp->num_pivots >= 2 && comp->num_pivots <= 9);
@@ -1771,6 +1773,13 @@ Index: FFmpeg/libavfilter/vf_tonemap_videotoolbox.m
+ return ret; + return ret;
+ } + }
+ +
+ if (ctx->trc != AVCOL_TRC_SMPTE2084) {
+ av_frame_side_data_remove(&outlink->side_data, &outlink->nb_side_data,
+ AV_FRAME_DATA_MASTERING_DISPLAY_METADATA);
+ av_frame_side_data_remove(&outlink->side_data, &outlink->nb_side_data,
+ AV_FRAME_DATA_CONTENT_LIGHT_LEVEL);
+ }
+
+ return 0; + return 0;
+} +}
+ +
@@ -1883,7 +1892,7 @@ Index: FFmpeg/libavfilter/vf_tonemap_videotoolbox.m
+ const AVDOVIRpuDataHeader *rpu = av_dovi_get_header(metadata); + const AVDOVIRpuDataHeader *rpu = av_dovi_get_header(metadata);
+ // only map dovi rpus that don't require an EL + // only map dovi rpus that don't require an EL
+ if (rpu->disable_residual_flag) { + if (rpu->disable_residual_flag) {
+ struct DoviMetadata *dovi = av_malloc(sizeof(*dovi)); + struct FFDOVIMetadataRemap *dovi = av_malloc(sizeof(*dovi));
+ ctx->dovi = dovi; + ctx->dovi = dovi;
+ if (!ctx->dovi) + if (!ctx->dovi)
+ goto fail; + goto fail;
@@ -2166,15 +2175,16 @@ Index: FFmpeg/libavfilter/vf_tonemap_videotoolbox.m
+ }, + },
+}; +};
+ +
+const AVFilter ff_vf_tonemap_videotoolbox = { +const FFFilter ff_vf_tonemap_videotoolbox = {
+ .name = "tonemap_videotoolbox", + .p.name = "tonemap_videotoolbox",
+ .description = NULL_IF_CONFIG_SMALL("Perform HDR to SDR conversion with Metal."), + .p.description = NULL_IF_CONFIG_SMALL("Perform HDR to SDR conversion with Metal."),
+ .priv_size = sizeof(TonemapVideoToolboxContext), + .priv_size = sizeof(TonemapVideoToolboxContext),
+ .priv_class = &tonemap_videotoolbox_class, + .p.priv_class = &tonemap_videotoolbox_class,
+ .preinit = tonemap_videotoolbox_preinit, + .preinit = tonemap_videotoolbox_preinit,
+ .uninit = tonemap_videotoolbox_uninit, + .uninit = tonemap_videotoolbox_uninit,
+ FILTER_INPUTS(tonemap_videotoolbox_inputs), + FILTER_INPUTS(tonemap_videotoolbox_inputs),
+ FILTER_OUTPUTS(tonemap_videotoolbox_outputs), + FILTER_OUTPUTS(tonemap_videotoolbox_outputs),
+ FILTER_QUERY_FUNC(tonemap_videotoolbox_query_formats), + FILTER_QUERY_FUNC(tonemap_videotoolbox_query_formats),
+ .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE, + .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE,
+ .p.flags = AVFILTER_FLAG_HWDEVICE,
+}; +};
@@ -1,68 +0,0 @@
Index: FFmpeg/libavfilter/vf_scale_vt.c
===================================================================
--- FFmpeg.orig/libavfilter/vf_scale_vt.c
+++ FFmpeg/libavfilter/vf_scale_vt.c
@@ -41,11 +41,26 @@ typedef struct ScaleVtContext {
enum AVColorPrimaries colour_primaries;
enum AVColorTransferCharacteristic colour_transfer;
enum AVColorSpace colour_matrix;
+ enum AVPixelFormat format;
char *colour_primaries_string;
char *colour_transfer_string;
char *colour_matrix_string;
} ScaleVtContext;
+static const enum AVPixelFormat supported_formats[] = {
+ AV_PIX_FMT_NV12,
+ AV_PIX_FMT_P010,
+ AV_PIX_FMT_NONE,
+};
+
+static int format_is_supported(enum AVPixelFormat fmt)
+{
+ for (int i = 0; i < FF_ARRAY_ELEMS(supported_formats); i++)
+ if (supported_formats[i] == fmt)
+ return 1;
+ return 0;
+}
+
static av_cold int scale_vt_init(AVFilterContext *avctx)
{
ScaleVtContext *s = avctx->priv;
@@ -182,6 +197,7 @@ static int scale_vt_config_output(AVFilt
FilterLink *inl = ff_filter_link(inlink);
AVHWFramesContext *hw_frame_ctx_in;
AVHWFramesContext *hw_frame_ctx_out;
+ enum AVPixelFormat out_format;
err = ff_scale_eval_dimensions(s, s->w_expr, s->h_expr, inlink, outlink,
&s->output_width,
@@ -201,11 +217,18 @@ static int scale_vt_config_output(AVFilt
hw_frame_ctx_in = (AVHWFramesContext *)inl->hw_frames_ctx->data;
+ out_format = (s->format == AV_PIX_FMT_NONE) ? hw_frame_ctx_in->sw_format : s->format;
+ if (!format_is_supported(s->format)) {
+ av_log(s, AV_LOG_ERROR, "Unsupported output format: %s\n",
+ av_get_pix_fmt_name(out_format));
+ return AVERROR(ENOSYS);
+ }
+
av_buffer_unref(&outl->hw_frames_ctx);
outl->hw_frames_ctx = av_hwframe_ctx_alloc(hw_frame_ctx_in->device_ref);
hw_frame_ctx_out = (AVHWFramesContext *)outl->hw_frames_ctx->data;
hw_frame_ctx_out->format = AV_PIX_FMT_VIDEOTOOLBOX;
- hw_frame_ctx_out->sw_format = hw_frame_ctx_in->sw_format;
+ hw_frame_ctx_out->sw_format = out_format;
hw_frame_ctx_out->width = outlink->w;
hw_frame_ctx_out->height = outlink->h;
((AVVTFramesContext *)hw_frame_ctx_out->hwctx)->color_range = ((AVVTFramesContext *)hw_frame_ctx_in->hwctx)->color_range;
@@ -238,6 +261,8 @@ static const AVOption scale_vt_options[]
OFFSET(colour_primaries_string), AV_OPT_TYPE_STRING, { .str = NULL }, .flags = FLAGS },
{ "color_transfer", "Output colour transfer characteristics",
OFFSET(colour_transfer_string), AV_OPT_TYPE_STRING, { .str = NULL }, .flags = FLAGS },
+ { "format", "Output pixel format",
+ OFFSET(format), AV_OPT_TYPE_PIXEL_FMT, { .i64 = AV_PIX_FMT_NONE }, AV_PIX_FMT_NONE, INT_MAX, .flags = FLAGS },
{ NULL },
};
@@ -2,7 +2,7 @@ Index: FFmpeg/configure
=================================================================== ===================================================================
--- FFmpeg.orig/configure --- FFmpeg.orig/configure
+++ FFmpeg/configure +++ FFmpeg/configure
@@ -3525,6 +3525,8 @@ h264_videotoolbox_encoder_deps="pthreads @@ -3767,6 +3767,8 @@ h264_videotoolbox_encoder_deps="pthreads
h264_videotoolbox_encoder_select="atsc_a53 videotoolbox_encoder" h264_videotoolbox_encoder_select="atsc_a53 videotoolbox_encoder"
hevc_videotoolbox_encoder_deps="pthreads" hevc_videotoolbox_encoder_deps="pthreads"
hevc_videotoolbox_encoder_select="atsc_a53 videotoolbox_encoder" hevc_videotoolbox_encoder_select="atsc_a53 videotoolbox_encoder"
@@ -15,31 +15,31 @@ Index: FFmpeg/libavcodec/Makefile
=================================================================== ===================================================================
--- FFmpeg.orig/libavcodec/Makefile --- FFmpeg.orig/libavcodec/Makefile
+++ FFmpeg/libavcodec/Makefile +++ FFmpeg/libavcodec/Makefile
@@ -518,6 +518,7 @@ OBJS-$(CONFIG_MJPEGB_DECODER) + @@ -544,6 +544,7 @@ OBJS-$(CONFIG_MJPEG_QSV_ENCODER) +
OBJS-$(CONFIG_MJPEG_CUVID_DECODER) += cuviddec.o
OBJS-$(CONFIG_MJPEG_QSV_ENCODER) += qsvenc_jpeg.o
OBJS-$(CONFIG_MJPEG_VAAPI_ENCODER) += vaapi_encode_mjpeg.o
+OBJS-$(CONFIG_MJPEG_VIDEOTOOLBOX_ENCODER) += videotoolboxenc.o
OBJS-$(CONFIG_MJPEG_RKMPP_DECODER) += rkmppdec.o OBJS-$(CONFIG_MJPEG_RKMPP_DECODER) += rkmppdec.o
OBJS-$(CONFIG_MJPEG_RKMPP_ENCODER) += rkmppenc.o OBJS-$(CONFIG_MJPEG_RKMPP_ENCODER) += rkmppenc.o
OBJS-$(CONFIG_MJPEG_VAAPI_ENCODER) += vaapi_encode_mjpeg.o
+OBJS-$(CONFIG_MJPEG_VIDEOTOOLBOX_ENCODER) += videotoolboxenc.o
OBJS-$(CONFIG_MLP_DECODER) += mlpdec.o mlpdsp.o OBJS-$(CONFIG_MLP_DECODER) += mlpdec.o mlpdsp.o
OBJS-$(CONFIG_MLP_ENCODER) += mlpenc.o mlp.o
OBJS-$(CONFIG_MMVIDEO_DECODER) += mmvideo.o
Index: FFmpeg/libavcodec/allcodecs.c Index: FFmpeg/libavcodec/allcodecs.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavcodec/allcodecs.c --- FFmpeg.orig/libavcodec/allcodecs.c
+++ FFmpeg/libavcodec/allcodecs.c +++ FFmpeg/libavcodec/allcodecs.c
@@ -875,6 +875,7 @@ extern const FFCodec ff_mjpeg_cuvid_deco @@ -916,6 +916,7 @@ extern const FFCodec ff_mjpeg_qsv_decode
extern const FFCodec ff_mjpeg_qsv_encoder;
extern const FFCodec ff_mjpeg_qsv_decoder;
extern const FFCodec ff_mjpeg_vaapi_encoder;
+extern const FFCodec ff_mjpeg_videotoolbox_encoder;
extern const FFCodec ff_mjpeg_rkmpp_decoder; extern const FFCodec ff_mjpeg_rkmpp_decoder;
extern const FFCodec ff_mjpeg_rkmpp_encoder; extern const FFCodec ff_mjpeg_rkmpp_encoder;
extern const FFCodec ff_mjpeg_vaapi_encoder;
+extern const FFCodec ff_mjpeg_videotoolbox_encoder;
extern const FFCodec ff_mp3_mediacodec_decoder; extern const FFCodec ff_mp3_mediacodec_decoder;
extern const FFCodec ff_mp3_mf_encoder;
extern const FFCodec ff_mpeg1_cuvid_decoder;
Index: FFmpeg/libavcodec/videotoolboxenc.c Index: FFmpeg/libavcodec/videotoolboxenc.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavcodec/videotoolboxenc.c --- FFmpeg.orig/libavcodec/videotoolboxenc.c
+++ FFmpeg/libavcodec/videotoolboxenc.c +++ FFmpeg/libavcodec/videotoolboxenc.c
@@ -546,6 +546,7 @@ static CMVideoCodecType get_cm_codec_typ @@ -556,6 +556,7 @@ static CMVideoCodecType get_cm_codec_typ
else else
return MKBETAG('a','p','c','n'); // kCMVideoCodecType_AppleProRes422 return MKBETAG('a','p','c','n'); // kCMVideoCodecType_AppleProRes422
} }
@@ -47,7 +47,7 @@ Index: FFmpeg/libavcodec/videotoolboxenc.c
default: return 0; default: return 0;
} }
} }
@@ -1243,7 +1244,7 @@ static int vtenc_create_encoder(AVCodecC @@ -1258,7 +1259,7 @@ static int vtenc_create_encoder(AVCodecC
kVTCompressionPropertyKey_Quality, kVTCompressionPropertyKey_Quality,
quality_num); quality_num);
CFRelease(quality_num); CFRelease(quality_num);
@@ -56,7 +56,7 @@ Index: FFmpeg/libavcodec/videotoolboxenc.c
bit_rate_num = CFNumberCreate(kCFAllocatorDefault, bit_rate_num = CFNumberCreate(kCFAllocatorDefault,
kCFNumberSInt32Type, kCFNumberSInt32Type,
&bit_rate); &bit_rate);
@@ -1357,7 +1358,7 @@ static int vtenc_create_encoder(AVCodecC @@ -1374,7 +1375,7 @@ static int vtenc_create_encoder(AVCodecC
} }
} }
@@ -65,7 +65,7 @@ Index: FFmpeg/libavcodec/videotoolboxenc.c
CFNumberRef interval = CFNumberCreate(kCFAllocatorDefault, CFNumberRef interval = CFNumberCreate(kCFAllocatorDefault,
kCFNumberIntType, kCFNumberIntType,
&avctx->gop_size); &avctx->gop_size);
@@ -1506,7 +1507,7 @@ static int vtenc_create_encoder(AVCodecC @@ -1523,7 +1524,7 @@ static int vtenc_create_encoder(AVCodecC
} }
} }
@@ -74,7 +74,7 @@ Index: FFmpeg/libavcodec/videotoolboxenc.c
status = VTSessionSetProperty(vtctx->session, status = VTSessionSetProperty(vtctx->session,
kVTCompressionPropertyKey_AllowFrameReordering, kVTCompressionPropertyKey_AllowFrameReordering,
kCFBooleanFalse); kCFBooleanFalse);
@@ -2875,6 +2876,13 @@ static const enum AVPixelFormat prores_p @@ -2769,6 +2770,13 @@ static const enum AVPixelFormat prores_p
AV_PIX_FMT_NONE AV_PIX_FMT_NONE
}; };
@@ -88,7 +88,7 @@ Index: FFmpeg/libavcodec/videotoolboxenc.c
#define VE AV_OPT_FLAG_VIDEO_PARAM | AV_OPT_FLAG_ENCODING_PARAM #define VE AV_OPT_FLAG_VIDEO_PARAM | AV_OPT_FLAG_ENCODING_PARAM
#define COMMON_OPTIONS \ #define COMMON_OPTIONS \
{ "allow_sw", "Allow software encoding", OFFSET(allow_sw), AV_OPT_TYPE_BOOL, \ { "allow_sw", "Allow software encoding", OFFSET(allow_sw), AV_OPT_TYPE_BOOL, \
@@ -3046,4 +3054,36 @@ const FFCodec ff_prores_videotoolbox_enc @@ -2944,4 +2952,36 @@ const FFCodec ff_prores_videotoolbox_enc
.caps_internal = FF_CODEC_CAP_INIT_CLEANUP, .caps_internal = FF_CODEC_CAP_INIT_CLEANUP,
.p.wrapper_name = "videotoolbox", .p.wrapper_name = "videotoolbox",
.hw_configs = vt_encode_hw_configs, .hw_configs = vt_encode_hw_configs,
@@ -114,7 +114,7 @@ Index: FFmpeg/libavcodec/videotoolboxenc.c
+ .p.capabilities = AV_CODEC_CAP_DR1 | AV_CODEC_CAP_DELAY | + .p.capabilities = AV_CODEC_CAP_DR1 | AV_CODEC_CAP_DELAY |
+ AV_CODEC_CAP_HARDWARE, + AV_CODEC_CAP_HARDWARE,
+ .priv_data_size = sizeof(VTEncContext), + .priv_data_size = sizeof(VTEncContext),
+ .p.pix_fmts = mjpeg_pix_fmts, + CODEC_PIXFMTS_ARRAY(mjpeg_pix_fmts),
+ .defaults = vt_defaults, + .defaults = vt_defaults,
+ .color_ranges = AVCOL_RANGE_MPEG | AVCOL_RANGE_JPEG, + .color_ranges = AVCOL_RANGE_MPEG | AVCOL_RANGE_JPEG,
+ .init = vtenc_init, + .init = vtenc_init,
@@ -2,7 +2,7 @@ Index: FFmpeg/libavcodec/avcodec.h
=================================================================== ===================================================================
--- FFmpeg.orig/libavcodec/avcodec.h --- FFmpeg.orig/libavcodec/avcodec.h
+++ FFmpeg/libavcodec/avcodec.h +++ FFmpeg/libavcodec/avcodec.h
@@ -2185,6 +2185,13 @@ typedef struct AVHWAccel { @@ -2041,6 +2041,13 @@ typedef struct AVHWAccel {
#define AV_HWACCEL_FLAG_UNSAFE_OUTPUT (1 << 3) #define AV_HWACCEL_FLAG_UNSAFE_OUTPUT (1 << 3)
/** /**
@@ -20,19 +20,19 @@ Index: FFmpeg/libavcodec/options_table.h
=================================================================== ===================================================================
--- FFmpeg.orig/libavcodec/options_table.h --- FFmpeg.orig/libavcodec/options_table.h
+++ FFmpeg/libavcodec/options_table.h +++ FFmpeg/libavcodec/options_table.h
@@ -412,6 +412,7 @@ static const AVOption avcodec_options[] @@ -401,6 +401,7 @@ static const AVOption avcodec_options[]
{"mastering_display_metadata", .default_val.i64 = AV_PKT_DATA_MASTERING_DISPLAY_METADATA, .type = AV_OPT_TYPE_CONST, .flags = A|D, .unit = "side_data_pkt" }, {"allow_high_depth", "allow to output YUV pixel formats with a different chroma sampling than 4:2:0 and/or other than 8 bits per component", 0, AV_OPT_TYPE_CONST, {.i64 = AV_HWACCEL_FLAG_ALLOW_HIGH_DEPTH }, INT_MIN, INT_MAX, V | D, .unit = "hwaccel_flags"},
{"content_light_level", .default_val.i64 = AV_PKT_DATA_CONTENT_LIGHT_LEVEL, .type = AV_OPT_TYPE_CONST, .flags = A|D, .unit = "side_data_pkt" }, {"allow_profile_mismatch", "attempt to decode anyway if HW accelerated decoder's supported profiles do not exactly match the stream", 0, AV_OPT_TYPE_CONST, {.i64 = AV_HWACCEL_FLAG_ALLOW_PROFILE_MISMATCH }, INT_MIN, INT_MAX, V | D, .unit = "hwaccel_flags"},
{"icc_profile", .default_val.i64 = AV_PKT_DATA_ICC_PROFILE, .type = AV_OPT_TYPE_CONST, .flags = A|D, .unit = "side_data_pkt" }, {"unsafe_output", "allow potentially unsafe hwaccel frame output that might require special care to process successfully", 0, AV_OPT_TYPE_CONST, {.i64 = AV_HWACCEL_FLAG_UNSAFE_OUTPUT }, INT_MIN, INT_MAX, V | D, .unit = "hwaccel_flags"},
+{"low_priority", "attempt to run decode pipeline at a lower priority than is used for realtime decoding", 0, AV_OPT_TYPE_CONST, {.i64 = AV_HWACCEL_FLAG_LOW_PRIORITY }, INT_MIN, INT_MAX, V | D, .unit = "hwaccel_flags"}, +{"low_priority", "attempt to run decode pipeline at a lower priority than is used for realtime decoding", 0, AV_OPT_TYPE_CONST, {.i64 = AV_HWACCEL_FLAG_LOW_PRIORITY }, INT_MIN, INT_MAX, V | D, .unit = "hwaccel_flags"},
{NULL}, {"extra_hw_frames", "Number of extra hardware frames to allocate for the user", OFFSET(extra_hw_frames), AV_OPT_TYPE_INT, { .i64 = -1 }, -1, INT_MAX, V|D },
}; {"discard_damaged_percentage", "Percentage of damaged samples to discard a frame", OFFSET(discard_damaged_percentage), AV_OPT_TYPE_INT, {.i64 = 95 }, 0, 100, V|D },
{"side_data_prefer_packet", "Comma-separated list of side data types for which user-supplied (container) data is preferred over coded bytestream",
Index: FFmpeg/libavcodec/videotoolbox.c Index: FFmpeg/libavcodec/videotoolbox.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavcodec/videotoolbox.c --- FFmpeg.orig/libavcodec/videotoolbox.c
+++ FFmpeg/libavcodec/videotoolbox.c +++ FFmpeg/libavcodec/videotoolbox.c
@@ -985,6 +985,23 @@ static int videotoolbox_start(AVCodecCon @@ -1025,6 +1025,23 @@ static int videotoolbox_start(AVCodecCon
av_log(avctx, AV_LOG_VERBOSE, "VideoToolbox reported invalid data.\n"); av_log(avctx, AV_LOG_VERBOSE, "VideoToolbox reported invalid data.\n");
return AVERROR_INVALIDDATA; return AVERROR_INVALIDDATA;
case 0: case 0:
@@ -2,7 +2,7 @@ Index: FFmpeg/configure
=================================================================== ===================================================================
--- FFmpeg.orig/configure --- FFmpeg.orig/configure
+++ FFmpeg/configure +++ FFmpeg/configure
@@ -3321,6 +3321,8 @@ thumbnail_cuda_filter_deps="ffnvcodec" @@ -3527,6 +3527,8 @@ thumbnail_cuda_filter_deps="ffnvcodec"
thumbnail_cuda_filter_deps_any="cuda_nvcc cuda_llvm" thumbnail_cuda_filter_deps_any="cuda_nvcc cuda_llvm"
tonemap_cuda_filter_deps="ffnvcodec const_nan" tonemap_cuda_filter_deps="ffnvcodec const_nan"
tonemap_cuda_filter_deps_any="cuda_nvcc cuda_llvm" tonemap_cuda_filter_deps_any="cuda_nvcc cuda_llvm"
@@ -15,7 +15,7 @@ Index: FFmpeg/libavfilter/Makefile
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/Makefile --- FFmpeg.orig/libavfilter/Makefile
+++ FFmpeg/libavfilter/Makefile +++ FFmpeg/libavfilter/Makefile
@@ -542,6 +542,8 @@ OBJS-$(CONFIG_TONEMAP_VIDEOTOOLBOX_FILTE @@ -559,6 +559,8 @@ OBJS-$(CONFIG_TONEMAP_VIDEOTOOLBOX_FILTE
metal/utils.o metal/utils.o
OBJS-$(CONFIG_TPAD_FILTER) += vf_tpad.o OBJS-$(CONFIG_TPAD_FILTER) += vf_tpad.o
OBJS-$(CONFIG_TRANSPOSE_FILTER) += vf_transpose.o OBJS-$(CONFIG_TRANSPOSE_FILTER) += vf_transpose.o
@@ -28,21 +28,21 @@ Index: FFmpeg/libavfilter/allfilters.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/allfilters.c --- FFmpeg.orig/libavfilter/allfilters.c
+++ FFmpeg/libavfilter/allfilters.c +++ FFmpeg/libavfilter/allfilters.c
@@ -505,6 +505,7 @@ extern const AVFilter ff_vf_tonemap_vaap @@ -520,6 +520,7 @@ extern const FFFilter ff_vf_tonemap_vaap
extern const AVFilter ff_vf_tonemap_videotoolbox; extern const FFFilter ff_vf_tonemap_videotoolbox;
extern const AVFilter ff_vf_tpad; extern const FFFilter ff_vf_tpad;
extern const AVFilter ff_vf_transpose; extern const FFFilter ff_vf_transpose;
+extern const AVFilter ff_vf_transpose_cuda; +extern const FFFilter ff_vf_transpose_cuda;
extern const AVFilter ff_vf_transpose_npp; extern const FFFilter ff_vf_transpose_npp;
extern const AVFilter ff_vf_transpose_opencl; extern const FFFilter ff_vf_transpose_opencl;
extern const AVFilter ff_vf_transpose_vaapi; extern const FFFilter ff_vf_transpose_vaapi;
Index: FFmpeg/libavfilter/vf_transpose_cuda.c Index: FFmpeg/libavfilter/vf_transpose_cuda.c
=================================================================== ===================================================================
--- /dev/null --- /dev/null
+++ FFmpeg/libavfilter/vf_transpose_cuda.c +++ FFmpeg/libavfilter/vf_transpose_cuda.c
@@ -0,0 +1,483 @@ @@ -0,0 +1,564 @@
+/* +/*
+ * Copyright (C) 2024 NyanMisaka + * Copyright (C) 2026 NyanMisaka
+ * + *
+ * This file is part of FFmpeg. + * This file is part of FFmpeg.
+ * + *
@@ -61,6 +61,7 @@ Index: FFmpeg/libavfilter/vf_transpose_cuda.c
+ * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA + * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
+ */ + */
+ +
+#include "libavutil/avassert.h"
+#include "libavutil/common.h" +#include "libavutil/common.h"
+#include "libavutil/hwcontext.h" +#include "libavutil/hwcontext.h"
+#include "libavutil/hwcontext_cuda_internal.h" +#include "libavutil/hwcontext_cuda_internal.h"
@@ -76,19 +77,26 @@ Index: FFmpeg/libavfilter/vf_transpose_cuda.c
+ +
+#include "cuda/load_helper.h" +#include "cuda/load_helper.h"
+ +
+#define CHECK_CU(x) FF_CUDA_CHECK_DL(ctx, s->hwctx->internal->cuda_dl, x)
+#define DIV_UP(a, b) ( ((a) + (b) - 1) / (b) ) +#define DIV_UP(a, b) ( ((a) + (b) - 1) / (b) )
+
+#define BLOCK_X 32 +#define BLOCK_X 32
+#define BLOCK_Y 16 +#define BLOCK_Y 16
+ +
+#define CHECK_CU(x) FF_CUDA_CHECK_DL(ctx, s->hwctx->internal->cuda_dl, x)
+
+static const enum AVPixelFormat supported_formats[] = { +static const enum AVPixelFormat supported_formats[] = {
+ AV_PIX_FMT_YUV420P, + AV_PIX_FMT_YUV420P,
+ AV_PIX_FMT_NV12, + AV_PIX_FMT_YUV422P,
+ AV_PIX_FMT_YUV444P, + AV_PIX_FMT_YUV444P,
+ AV_PIX_FMT_P010, + AV_PIX_FMT_YUV420P10,
+ AV_PIX_FMT_P016, + AV_PIX_FMT_YUV422P10,
+ AV_PIX_FMT_YUV444P10,
+ AV_PIX_FMT_YUV444P16, + AV_PIX_FMT_YUV444P16,
+ AV_PIX_FMT_NV12,
+ AV_PIX_FMT_NV16,
+ AV_PIX_FMT_P010,
+ AV_PIX_FMT_P210,
+ AV_PIX_FMT_P016,
+ AV_PIX_FMT_P216,
+ AV_PIX_FMT_0RGB32, + AV_PIX_FMT_0RGB32,
+ AV_PIX_FMT_0BGR32, + AV_PIX_FMT_0BGR32,
+ AV_PIX_FMT_RGB32, + AV_PIX_FMT_RGB32,
@@ -109,6 +117,9 @@ Index: FFmpeg/libavfilter/vf_transpose_cuda.c
+ CUmodule cu_module; + CUmodule cu_module;
+ CUfunction cu_func_uchar; + CUfunction cu_func_uchar;
+ CUfunction cu_func_ushort; + CUfunction cu_func_ushort;
+ CUfunction cu_func_uchar2;
+ CUfunction cu_func_ushort2;
+ CUfunction cu_func_uchar4;
+ CUstream cu_stream; + CUstream cu_stream;
+ +
+ int flip_wh; + int flip_wh;
@@ -190,12 +201,9 @@ Index: FFmpeg/libavfilter/vf_transpose_cuda.c
+ +
+static int format_is_supported(enum AVPixelFormat fmt) +static int format_is_supported(enum AVPixelFormat fmt)
+{ +{
+ int i; + for (int i = 0; i < FF_ARRAY_ELEMS(supported_formats); i++)
+
+ for (i = 0; i < FF_ARRAY_ELEMS(supported_formats); i++)
+ if (supported_formats[i] == fmt) + if (supported_formats[i] == fmt)
+ return 1; + return 1;
+
+ return 0; + return 0;
+} +}
+ +
@@ -306,13 +314,23 @@ Index: FFmpeg/libavfilter/vf_transpose_cuda.c
+ if (ret < 0) + if (ret < 0)
+ goto exit; + goto exit;
+ +
+ ret = CHECK_CU(cu->cuModuleGetFunction(&s->cu_func_uchar, + ret = CHECK_CU(cu->cuModuleGetFunction(&s->cu_func_uchar, s->cu_module, "Transpose_Cuda_uchar"));
+ s->cu_module, "Transpose_Cuda_uchar"));
+ if (ret < 0) + if (ret < 0)
+ goto exit; + goto exit;
+ +
+ ret = CHECK_CU(cu->cuModuleGetFunction(&s->cu_func_ushort, + ret = CHECK_CU(cu->cuModuleGetFunction(&s->cu_func_ushort, s->cu_module, "Transpose_Cuda_ushort"));
+ s->cu_module, "Transpose_Cuda_ushort")); + if (ret < 0)
+ goto exit;
+
+ ret = CHECK_CU(cu->cuModuleGetFunction(&s->cu_func_uchar2, s->cu_module, "Transpose_Cuda_uchar2"));
+ if (ret < 0)
+ goto exit;
+
+ ret = CHECK_CU(cu->cuModuleGetFunction(&s->cu_func_ushort2, s->cu_module, "Transpose_Cuda_ushort2"));
+ if (ret < 0)
+ goto exit;
+
+ ret = CHECK_CU(cu->cuModuleGetFunction(&s->cu_func_uchar4, s->cu_module, "Transpose_Cuda_uchar4"));
+ if (ret < 0) + if (ret < 0)
+ goto exit; + goto exit;
+ +
@@ -325,61 +343,126 @@ Index: FFmpeg/libavfilter/vf_transpose_cuda.c
+ return ret; + return ret;
+} +}
+ +
+static int call_transpose_kernel(AVFilterContext *ctx, +static CUresult call_kernel(AVFilterContext *ctx,
+ int pix_size, + CUfunction cu_func,
+ int pix_step, + CUarray_format cu_format,
+ int pix_offset, + int channels,
+ CUdeviceptr dst_data, + int is_422_uv, // Dst* & Src* are 4:2:2 UV planes
+ int dst_width, + CUdeviceptr dst0,
+ int dst_height, + CUdeviceptr dst1, // Dst1 is for fully planar V, optional
+ int dst_pitch, + int dst_width, // Width is pixels per channel
+ CUdeviceptr src_data, + int dst_height, // Height is pixels per channel
+ int src_width, + int dst_pitch, // Pitch is elements per channel
+ int src_height, + CUdeviceptr src0,
+ int src_pitch) + CUdeviceptr src1, // Src1 is for fully planar V, optional
+ int src_width, // Width is pixels per channel
+ int src_height, // Height is pixels per channel
+ int src_pitch)
+{ +{
+ TransposeCUDAContext *s = ctx->priv; + TransposeCUDAContext *s = ctx->priv;
+ CudaFunctions *cu = s->hwctx->internal->cuda_dl; + CudaFunctions *cu = s->hwctx->internal->cuda_dl;
+ CUtexObject src0_tex = 0, src1_tex = 0;
+ int ret;
+ +
+ void* kernel_args[] = { + void *kernel_args[] = {
+ &dst_data, &dst_width, &dst_height, &dst_pitch, + &dst0, &dst1, &dst_width, &dst_height, &dst_pitch,
+ &src_data, &src_width, &src_height, &src_pitch, + &src0_tex, &src1_tex, &s->dir,
+ &pix_step, &pix_offset, &s->dir,
+ }; + };
+ +
+ return CHECK_CU(cu->cuLaunchKernel(pix_size == 1 ? s->cu_func_uchar : s->cu_func_ushort, + CUDA_TEXTURE_DESC tex_desc = {
+ DIV_UP(dst_width, BLOCK_X), DIV_UP(dst_height, BLOCK_Y), 1, + .addressMode = { CU_TR_ADDRESS_MODE_CLAMP,
+ BLOCK_X, BLOCK_Y, 1, 0, s->cu_stream, kernel_args, NULL)); + CU_TR_ADDRESS_MODE_CLAMP },
+ .filterMode = is_422_uv ? CU_TR_FILTER_MODE_LINEAR
+ : CU_TR_FILTER_MODE_POINT,
+ .flags = 2 /* CU_TRSF_NORMALIZED_COORDINATES */
+ };
+ CUDA_RESOURCE_DESC res_desc = {
+ .resType = CU_RESOURCE_TYPE_PITCH2D,
+ .res.pitch2D.format = cu_format,
+ .res.pitch2D.numChannels = channels,
+ .res.pitch2D.pitchInBytes = src_pitch,
+ .res.pitch2D.width = src_width,
+ .res.pitch2D.height = src_height
+ };
+
+ res_desc.res.pitch2D.devPtr = (CUdeviceptr)src0;
+ ret = CHECK_CU(cu->cuTexObjectCreate(&src0_tex, &res_desc, &tex_desc, NULL));
+ if (ret < 0)
+ goto exit;
+
+ if (src1) {
+ res_desc.res.pitch2D.devPtr = (CUdeviceptr)src1;
+ ret = CHECK_CU(cu->cuTexObjectCreate(&src1_tex, &res_desc, &tex_desc, NULL));
+ if (ret < 0)
+ goto exit;
+ }
+
+ ret = CHECK_CU(cu->cuLaunchKernel(cu_func,
+ DIV_UP(dst_width, BLOCK_X), DIV_UP(dst_height, BLOCK_Y), 1,
+ BLOCK_X, BLOCK_Y, 1, 0, s->cu_stream, kernel_args, NULL));
+exit:
+ if (src0_tex)
+ CHECK_CU(cu->cuTexObjectDestroy(src0_tex));
+ if (src1_tex)
+ CHECK_CU(cu->cuTexObjectDestroy(src1_tex));
+
+ return ret;
+} +}
+ +
+static int cudatranspose_rotate(AVFilterContext *ctx, +static int cudatranspose_rotate(AVFilterContext *ctx,
+ AVFrame *out, AVFrame *in) + AVFrame *out, AVFrame *in)
+{ +{
+ TransposeCUDAContext *s = ctx->priv; + TransposeCUDAContext *s = ctx->priv;
+ int pix_size, pix_step, pix_offset; + int ret;
+ int c, p, ret;
+ +
+ for (c = 0; c < s->pix_desc->nb_components; c++) { + for (int c = 0; c < s->pix_desc->nb_components; c++) {
+ p = s->pix_desc->comp[c].plane; + const AVComponentDescriptor *comp = &s->pix_desc->comp[c];
+ if (!out->data[p] || !in->data[p]) + const int p = comp->plane;
+ return AVERROR(ENOMEM); + int pix_size, channels;
+ int is_planar_u, is_planar_v, is_422_uv;
+ CUfunction func;
+ CUarray_format format;
+ +
+ pix_size = (s->pix_desc->comp[c].depth + s->pix_desc->comp[c].shift) / 8; + pix_size = (comp->depth + 7) / 8;
+ if (!(pix_size == 1 || pix_size == 2)) + channels = comp->step / pix_size;
+ return AVERROR(ENOSYS); + if (pix_size > 2 || channels > 4)
+ av_unreachable("Unsupported pixel format!");
+ +
+ pix_step = s->pix_desc->comp[c].step / pix_size; + is_planar_u = p == 1 && channels == 1;
+ pix_offset = s->pix_desc->comp[c].offset / pix_size; + is_planar_v = p == 2 && channels == 1;
+ is_422_uv = p && s->pix_desc->log2_chroma_w == 1 && !s->pix_desc->log2_chroma_h;
+ +
+ ret = call_transpose_kernel(ctx, pix_size, pix_step, pix_offset, + if (comp->plane < c || is_planar_v) {
+ (CUdeviceptr)out->data[p], + // We process planes as a whole, so don't reprocess
+ AV_CEIL_RSHIFT(out->width, p ? s->pix_desc->log2_chroma_w : 0), + // them for additional components
+ AV_CEIL_RSHIFT(out->height, p ? s->pix_desc->log2_chroma_h : 0), + continue;
+ out->linesize[p] / pix_size, + }
+ (CUdeviceptr)in->data[p], +
+ AV_CEIL_RSHIFT(in->width, p ? s->pix_desc->log2_chroma_w : 0), + switch (pix_size) {
+ AV_CEIL_RSHIFT(in->height, p ? s->pix_desc->log2_chroma_h : 0), + case 1:
+ in->linesize[p] / pix_size); + func = channels == 4 ? s->cu_func_uchar4 :
+ channels == 2 ? s->cu_func_uchar2 : s->cu_func_uchar;
+ format = CU_AD_FORMAT_UNSIGNED_INT8;
+ break;
+ case 2:
+ func = channels == 2 ? s->cu_func_ushort2 : s->cu_func_ushort;
+ format = CU_AD_FORMAT_UNSIGNED_INT16;
+ break;
+ default:
+ av_unreachable("Unsupported pixel format!");
+ }
+
+ ret = call_kernel(ctx, func, format, channels, is_422_uv,
+ (CUdeviceptr)out->data[p],
+ (CUdeviceptr)(is_planar_u ? out->data[p+1] : NULL),
+ AV_CEIL_RSHIFT(out->width, p ? s->pix_desc->log2_chroma_w : 0),
+ AV_CEIL_RSHIFT(out->height, p ? s->pix_desc->log2_chroma_h : 0),
+ out->linesize[p] / comp->step,
+ (CUdeviceptr)in->data[p],
+ (CUdeviceptr)(is_planar_u ? in->data[p+1] : NULL),
+ AV_CEIL_RSHIFT(in->width, p ? s->pix_desc->log2_chroma_w : 0),
+ AV_CEIL_RSHIFT(in->height, p ? s->pix_desc->log2_chroma_h : 0),
+ in->linesize[p]);
+ if (ret < 0) + if (ret < 0)
+ return ret; + return ret;
+ } + }
@@ -387,20 +470,18 @@ Index: FFmpeg/libavfilter/vf_transpose_cuda.c
+ return 0; + return 0;
+} +}
+ +
+static int cudatranspose_filter_frame_internal(AVFilterContext *ctx, +static int cudatranspose_transpose(AVFilterContext *ctx,
+ AVFrame *out, AVFrame *in) + AVFrame *out, AVFrame *in)
+{ +{
+ TransposeCUDAContext *s = ctx->priv; + TransposeCUDAContext *s = ctx->priv;
+ AVFilterLink *outlink = ctx->outputs[0]; + AVFilterLink *outlink = ctx->outputs[0];
+ AVFrame *src = in;
+ int ret; + int ret;
+ +
+ ret = cudatranspose_rotate(ctx, s->frame, src); + ret = cudatranspose_rotate(ctx, s->frame, in);
+ if (ret < 0) + if (ret < 0)
+ return ret; + return ret;
+ +
+ src = s->frame; + ret = av_hwframe_get_buffer(s->frame->hw_frames_ctx, s->tmp_frame, 0);
+ ret = av_hwframe_get_buffer(src->hw_frames_ctx, s->tmp_frame, 0);
+ if (ret < 0) + if (ret < 0)
+ return ret; + return ret;
+ +
@@ -447,7 +528,7 @@ Index: FFmpeg/libavfilter/vf_transpose_cuda.c
+ if (ret < 0) + if (ret < 0)
+ goto fail; + goto fail;
+ +
+ ret = cudatranspose_filter_frame_internal(ctx, out, in); + ret = cudatranspose_transpose(ctx, out, in);
+ +
+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); + CHECK_CU(cu->cuCtxPopCurrent(&dummy));
+ if (ret < 0) + if (ret < 0)
@@ -497,9 +578,9 @@ Index: FFmpeg/libavfilter/vf_transpose_cuda.c
+ +
+static const AVFilterPad cudatranspose_inputs[] = { +static const AVFilterPad cudatranspose_inputs[] = {
+ { + {
+ .name = "default", + .name = "default",
+ .type = AVMEDIA_TYPE_VIDEO, + .type = AVMEDIA_TYPE_VIDEO,
+ .filter_frame = cudatranspose_filter_frame, + .filter_frame = cudatranspose_filter_frame,
+ .get_buffer.video = cudatranspose_get_video_buffer, + .get_buffer.video = cudatranspose_get_video_buffer,
+ }, + },
+}; +};
@@ -512,13 +593,13 @@ Index: FFmpeg/libavfilter/vf_transpose_cuda.c
+ }, + },
+}; +};
+ +
+const AVFilter ff_vf_transpose_cuda = { +const FFFilter ff_vf_transpose_cuda = {
+ .name = "transpose_cuda", + .p.name = "transpose_cuda",
+ .description = NULL_IF_CONFIG_SMALL("Transpose input video using CUDA"), + .p.description = NULL_IF_CONFIG_SMALL("Transpose input video using CUDA"),
+ .p.priv_class = &cudatranspose_class,
+ .init = cudatranspose_init, + .init = cudatranspose_init,
+ .uninit = cudatranspose_uninit, + .uninit = cudatranspose_uninit,
+ .priv_size = sizeof(TransposeCUDAContext), + .priv_size = sizeof(TransposeCUDAContext),
+ .priv_class = &cudatranspose_class,
+ FILTER_INPUTS(cudatranspose_inputs), + FILTER_INPUTS(cudatranspose_inputs),
+ FILTER_OUTPUTS(cudatranspose_outputs), + FILTER_OUTPUTS(cudatranspose_outputs),
+ FILTER_SINGLE_PIXFMT(AV_PIX_FMT_CUDA), + FILTER_SINGLE_PIXFMT(AV_PIX_FMT_CUDA),
@@ -528,9 +609,9 @@ Index: FFmpeg/libavfilter/vf_transpose_cuda.cu
=================================================================== ===================================================================
--- /dev/null --- /dev/null
+++ FFmpeg/libavfilter/vf_transpose_cuda.cu +++ FFmpeg/libavfilter/vf_transpose_cuda.cu
@@ -0,0 +1,61 @@ @@ -0,0 +1,85 @@
+/* +/*
+ * Copyright (C) 2024 NyanMisaka + * Copyright (C) 2026 NyanMisaka
+ * + *
+ * This file is part of FFmpeg. + * This file is part of FFmpeg.
+ * + *
@@ -549,44 +630,68 @@ Index: FFmpeg/libavfilter/vf_transpose_cuda.cu
+ * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA + * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
+ */ + */
+ +
+template<typename T> +#include "cuda/vector_helpers.cuh"
+__inline__ __device__ void transpose_func( +
+ T* dst, int dst_width, int dst_height, int dst_pitch, +__inline__ __device__ void map_input_ncoords(float *xi, float *yi,
+ T* src, int src_width, int src_height, int src_pitch, + int xo, int yo,
+ int pix_step, int pix_offset, int dir) + int dst_width, int dst_height,
+ int dir)
+{
+ int flip_wh = dir < 4;
+ *xi = flip_wh ? ((dir & 2) ? (dst_height - 1 - yo) : yo)
+ : ((dir == 6) ? xo : (dst_width - 1 - xo));
+ *yi = flip_wh ? ((dir & 1) ? (dst_width - 1 - xo) : xo)
+ : ((dir == 5) ? yo : (dst_height - 1 - yo));
+
+ *xi = (*xi + 0.5f) / (flip_wh ? dst_height : dst_width);
+ *yi = (*yi + 0.5f) / (flip_wh ? dst_width : dst_height);
+}
+
+template<typename T, int DST1, int FACTOR>
+__inline__ __device__ void Transpose_Cuda(
+ T *dst0, T *dst1, int dst_width, int dst_height, int dst_pitch,
+ cudaTextureObject_t src0_tex, cudaTextureObject_t src1_tex, int dir)
+{ +{
+ int xo = blockIdx.x * blockDim.x + threadIdx.x; + int xo = blockIdx.x * blockDim.x + threadIdx.x;
+ int yo = blockIdx.y * blockDim.y + threadIdx.y; + int yo = blockIdx.y * blockDim.y + threadIdx.y;
+ if (xo >= dst_width || yo >= dst_height) + if (xo >= dst_width || yo >= dst_height)
+ return; + return;
+ +
+ int xi = (dir < 4) ? ((dir & 2) ? (dst_height - 1 - yo) : yo) + float xi, yi;
+ : ((dir == 6) ? xo : (dst_width - 1 - xo)); + map_input_ncoords(&xi, &yi, xo, yo,
+ int yi = (dir < 4) ? ((dir & 1) ? (dst_width - 1 - xo) : xo) + dst_width, dst_height, dir);
+ : ((dir == 5) ? yo : (dst_height - 1 - yo));
+ if (xi >= src_width || yi >= src_height)
+ return;
+ +
+ int dst_pos = xo*pix_step + yo*dst_pitch + pix_offset; + dst0[yo*dst_pitch+xo] = from_floatN<T, floatT>(
+ int src_pos = xi*pix_step + yi*src_pitch + pix_offset; + saturate_rintf<floatT>(
+ dst[dst_pos] = src[src_pos]; + tex2D<floatT>(src0_tex, xi, yi), FACTOR
+ )
+ );
+ if (DST1 && dst1 && src1_tex) {
+ dst1[yo*dst_pitch+xo] = from_floatN<T, floatT>(
+ saturate_rintf<floatT>(
+ tex2D<floatT>(src1_tex, xi, yi), FACTOR
+ )
+ );
+ }
+} +}
+ +
+extern "C" { +extern "C" {
+ +
+#define TRANSPOSE_VARIANT(NAME, TYPE) \ +#define TRANSPOSE_KERNEL(NAME, TYPE, DST1, FACTOR) \
+__global__ void Transpose_Cuda_ ## NAME( \ +__global__ void Transpose_Cuda_ ## NAME( \
+ TYPE* dst, int dst_width, int dst_height, int dst_pitch, \ + TYPE *dst0, TYPE *dst1, int dst_width, int dst_height, int dst_pitch, \
+ TYPE* src, int src_width, int src_height, int src_pitch, \ + cudaTextureObject_t src0_tex, cudaTextureObject_t src1_tex, int dir) \
+ int pix_step, int pix_offset, int dir) \ +{ \
+{ \ + Transpose_Cuda<TYPE, DST1, FACTOR>( \
+ transpose_func( \ + dst0, dst1, dst_width, dst_height, dst_pitch, \
+ dst, dst_width, dst_height, dst_pitch, \ + src0_tex, src1_tex, dir \
+ src, src_width, src_height, src_pitch, \ + ); \
+ pix_step, pix_offset, dir); \
+} +}
+ +
+TRANSPOSE_VARIANT(uchar, unsigned char) +TRANSPOSE_KERNEL(uchar, uchar, 1, 0xFF)
+TRANSPOSE_VARIANT(ushort, unsigned short) +TRANSPOSE_KERNEL(ushort, ushort, 1, 0xFFFF)
+TRANSPOSE_KERNEL(uchar2, uchar2, 0, 0xFF)
+TRANSPOSE_KERNEL(ushort2, ushort2, 0, 0xFFFF)
+TRANSPOSE_KERNEL(uchar4, uchar4, 0, 0xFF)
+ +
+} /* extern "C" */ +} /* extern "C" */
@@ -8,10 +8,10 @@ Index: FFmpeg/libavfilter/opencl/transpose.cl
- int xin = (dir & 2) ? (size.y - 1 - y) : y; - int xin = (dir & 2) ? (size.y - 1 - y) : y;
- int yin = (dir & 1) ? (size.x - 1 - x) : x; - int yin = (dir & 1) ? (size.x - 1 - x) : x;
+ int xin = (dir < 4) ? ((dir & 2) ? (size.y - 1 - y) : y) + int xin = dir < 4 ? ((dir & 2) ? (size.y - 1 - y) : y)
+ : ((dir == 6) ? x : (size.x - 1 - x)); + : ((dir == 6) ? x : (size.x - 1 - x));
+ int yin = (dir < 4) ? ((dir & 1) ? (size.x - 1 - x) : x) + int yin = dir < 4 ? ((dir & 1) ? (size.x - 1 - x) : x)
+ : ((dir == 5) ? y : (size.y - 1 - y)); + : ((dir == 5) ? y : (size.y - 1 - y));
float4 data = read_imagef(src, sampler, (int2)(xin, yin)); float4 data = read_imagef(src, sampler, (int2)(xin, yin));
if (x < size.x && y < size.y) if (x < size.x && y < size.y)
@@ -2,7 +2,7 @@ Index: FFmpeg/configure
=================================================================== ===================================================================
--- FFmpeg.orig/configure --- FFmpeg.orig/configure
+++ FFmpeg/configure +++ FFmpeg/configure
@@ -2319,6 +2319,9 @@ HEADERS_LIST=" @@ -2458,6 +2458,9 @@ HEADERS_LIST="
INTRINSICS_LIST=" INTRINSICS_LIST="
intrinsics_neon intrinsics_neon
intrinsics_sse2 intrinsics_sse2
@@ -12,7 +12,7 @@ Index: FFmpeg/configure
" "
MATH_FUNCS=" MATH_FUNCS="
@@ -2753,6 +2756,9 @@ armv8_deps="aarch64" @@ -2915,6 +2918,9 @@ armv8_deps="aarch64"
neon_deps_any="aarch64 arm" neon_deps_any="aarch64 arm"
intrinsics_neon_deps="neon" intrinsics_neon_deps="neon"
intrinsics_sse2_deps="sse2" intrinsics_sse2_deps="sse2"
@@ -22,7 +22,7 @@ Index: FFmpeg/configure
vfp_deps="arm" vfp_deps="arm"
vfpv3_deps="vfp" vfpv3_deps="vfp"
setend_deps="arm" setend_deps="arm"
@@ -3984,6 +3990,7 @@ tinterlace_filter_deps="gpl" @@ -4243,6 +4249,7 @@ tinterlace_filter_deps="gpl"
tinterlace_merge_test_deps="tinterlace_filter" tinterlace_merge_test_deps="tinterlace_filter"
tinterlace_pad_test_deps="tinterlace_filter" tinterlace_pad_test_deps="tinterlace_filter"
tonemap_filter_deps="const_nan" tonemap_filter_deps="const_nan"
@@ -30,28 +30,28 @@ Index: FFmpeg/configure
tonemap_vaapi_filter_deps="vaapi VAProcFilterParameterBufferHDRToneMapping" tonemap_vaapi_filter_deps="vaapi VAProcFilterParameterBufferHDRToneMapping"
tonemap_videotoolbox_filter_deps="metal corevideo videotoolbox const_nan" tonemap_videotoolbox_filter_deps="metal corevideo videotoolbox const_nan"
tonemap_opencl_filter_deps="opencl const_nan" tonemap_opencl_filter_deps="opencl const_nan"
@@ -6529,6 +6536,19 @@ fi @@ -6835,6 +6842,19 @@ EOF
check_cc intrinsics_neon arm_neon.h "int16x8_t test = vdupq_n_s16(0)"
check_cc intrinsics_sse2 emmintrin.h "__m128i test = _mm_setzero_si128()"
+disable intrinsics_sse42 && test_cc -msse4.2 <<EOF && enable intrinsics_sse42 check_cc intrinsics_sse2 emmintrin.h "__m128i test = _mm_setzero_si128()"
+ disable intrinsics_sse42 && test_cc -msse4.2 <<EOF && enable intrinsics_sse42
+#include <immintrin.h> +#include <immintrin.h>
+int main(void) { __m128i t = _mm_cmpgt_epi64(_mm_setzero_si128(), _mm_setzero_si128()); return 0; } +int main(void) { __m128i t = _mm_cmpgt_epi64(_mm_setzero_si128(), _mm_setzero_si128()); return 0; }
+EOF +EOF
+disable intrinsics_fma3 && test_cc -mfma <<EOF && enable intrinsics_fma3 + disable intrinsics_fma3 && test_cc -mfma <<EOF && enable intrinsics_fma3
+#include <immintrin.h> +#include <immintrin.h>
+int main(void) { __m256 t = _mm256_fmadd_ps(_mm256_setzero_ps(), _mm256_setzero_ps(), _mm256_setzero_ps()); return 0; } +int main(void) { __m256 t = _mm256_fmadd_ps(_mm256_setzero_ps(), _mm256_setzero_ps(), _mm256_setzero_ps()); return 0; }
+EOF +EOF
+disable intrinsics_avx2 && test_cc -mavx2 <<EOF && enable intrinsics_avx2 + disable intrinsics_avx2 && test_cc -mavx2 <<EOF && enable intrinsics_avx2
+#include <immintrin.h> +#include <immintrin.h>
+int main(void) { __m256i t = _mm256_abs_epi32(_mm256_setzero_si256()); return 0; } +int main(void) { __m256i t = _mm256_abs_epi32(_mm256_setzero_si256()); return 0; }
+EOF +EOF
+ +
check_ldflags -Wl,--as-needed elif enabled loongarch; then
check_ldflags -Wl,-z,noexecstack enabled lsx && check_inline_asm lsx '"vadd.b $vr0, $vr1, $vr2"' '-mlsx' && append LSXFLAGS '-mlsx'
if [ $target_os = "darwin" ]; then enabled lasx && check_inline_asm lasx '"xvadd.b $xr0, $xr1, $xr2"' '-mlasx' && append LASXFLAGS '-mlasx'
@@ -7661,6 +7681,16 @@ elif enabled gcc; then @@ -8112,6 +8132,16 @@ elif enabled gcc; then
check_cflags -mpreferred-stack-boundary=4 check_allcflags -mpreferred-stack-boundary=4
;; ;;
esac esac
+ elif enabled x86_64; then + elif enabled x86_64; then
@@ -66,12 +66,12 @@ Index: FFmpeg/configure
+ esac + esac
fi fi
elif enabled llvm_gcc; then elif enabled llvm_gcc; then
check_cflags -mllvm -stack-alignment=16 check_allcflags -mllvm -stack-alignment=16
Index: FFmpeg/libavfilter/Makefile Index: FFmpeg/libavfilter/Makefile
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/Makefile --- FFmpeg.orig/libavfilter/Makefile
+++ FFmpeg/libavfilter/Makefile +++ FFmpeg/libavfilter/Makefile
@@ -532,6 +532,7 @@ OBJS-$(CONFIG_TMEDIAN_FILTER) @@ -549,6 +549,7 @@ OBJS-$(CONFIG_TMEDIAN_FILTER)
OBJS-$(CONFIG_TMIDEQUALIZER_FILTER) += vf_tmidequalizer.o OBJS-$(CONFIG_TMIDEQUALIZER_FILTER) += vf_tmidequalizer.o
OBJS-$(CONFIG_TMIX_FILTER) += vf_mix.o framesync.o OBJS-$(CONFIG_TMIX_FILTER) += vf_mix.o framesync.o
OBJS-$(CONFIG_TONEMAP_FILTER) += vf_tonemap.o OBJS-$(CONFIG_TONEMAP_FILTER) += vf_tonemap.o
@@ -83,13 +83,14 @@ Index: FFmpeg/libavfilter/aarch64/Makefile
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/aarch64/Makefile --- FFmpeg.orig/libavfilter/aarch64/Makefile
+++ FFmpeg/libavfilter/aarch64/Makefile +++ FFmpeg/libavfilter/aarch64/Makefile
@@ -1,5 +1,6 @@ @@ -1,6 +1,7 @@
OBJS-$(CONFIG_BWDIF_FILTER) += aarch64/vf_bwdif_init_aarch64.o OBJS-$(CONFIG_BWDIF_FILTER) += aarch64/vf_bwdif_init_aarch64.o
OBJS-$(CONFIG_COLORDETECT_FILTER) += aarch64/vf_colordetect_init.o
OBJS-$(CONFIG_NLMEANS_FILTER) += aarch64/vf_nlmeans_init.o OBJS-$(CONFIG_NLMEANS_FILTER) += aarch64/vf_nlmeans_init.o
+OBJS-$(CONFIG_TONEMAPX_FILTER) += aarch64/vf_tonemapx_intrin_neon.o +OBJS-$(CONFIG_TONEMAPX_FILTER) += aarch64/vf_tonemapx_intrin_neon.o
NEON-OBJS-$(CONFIG_BWDIF_FILTER) += aarch64/vf_bwdif_neon.o NEON-OBJS-$(CONFIG_BWDIF_FILTER) += aarch64/vf_bwdif_neon.o
NEON-OBJS-$(CONFIG_NLMEANS_FILTER) += aarch64/vf_nlmeans_neon.o NEON-OBJS-$(CONFIG_COLORDETECT_FILTER) += aarch64/vf_colordetect_neon.o
Index: FFmpeg/libavfilter/aarch64/vf_tonemapx_intrin_neon.c Index: FFmpeg/libavfilter/aarch64/vf_tonemapx_intrin_neon.c
=================================================================== ===================================================================
--- /dev/null --- /dev/null
@@ -2539,14 +2540,14 @@ Index: FFmpeg/libavfilter/allfilters.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/allfilters.c --- FFmpeg.orig/libavfilter/allfilters.c
+++ FFmpeg/libavfilter/allfilters.c +++ FFmpeg/libavfilter/allfilters.c
@@ -499,6 +499,7 @@ extern const AVFilter ff_vf_tmedian; @@ -514,6 +514,7 @@ extern const FFFilter ff_vf_tmedian;
extern const AVFilter ff_vf_tmidequalizer; extern const FFFilter ff_vf_tmidequalizer;
extern const AVFilter ff_vf_tmix; extern const FFFilter ff_vf_tmix;
extern const AVFilter ff_vf_tonemap; extern const FFFilter ff_vf_tonemap;
+extern const AVFilter ff_vf_tonemapx; +extern const FFFilter ff_vf_tonemapx;
extern const AVFilter ff_vf_tonemap_cuda; extern const FFFilter ff_vf_tonemap_cuda;
extern const AVFilter ff_vf_tonemap_opencl; extern const FFFilter ff_vf_tonemap_opencl;
extern const AVFilter ff_vf_tonemap_vaapi; extern const FFFilter ff_vf_tonemap_vaapi;
Index: FFmpeg/libavfilter/colorspace.c Index: FFmpeg/libavfilter/colorspace.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/colorspace.c --- FFmpeg.orig/libavfilter/colorspace.c
@@ -2559,8 +2560,8 @@ Index: FFmpeg/libavfilter/colorspace.c
#include "libavutil/frame.h" #include "libavutil/frame.h"
#include "libavutil/mastering_display_metadata.h" #include "libavutil/mastering_display_metadata.h"
#include "libavutil/pixdesc.h" #include "libavutil/pixdesc.h"
@@ -359,3 +360,51 @@ float inverse_eotf_arib_b67(float x) { @@ -359,3 +360,51 @@ float ff_inverse_eotf_arib_b67(float x)
float inverse_eotf_bt1886(float x) { float ff_inverse_eotf_bt1886(float x) {
return x > 0.0f ? powf(x, 1.0f / 2.4f) : 0.0f; return x > 0.0f ? powf(x, 1.0f / 2.4f) : 0.0f;
} }
+ +
@@ -2615,20 +2616,21 @@ Index: FFmpeg/libavfilter/colorspace.h
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/colorspace.h --- FFmpeg.orig/libavfilter/colorspace.h
+++ FFmpeg/libavfilter/colorspace.h +++ FFmpeg/libavfilter/colorspace.h
@@ -110,4 +110,8 @@ float eotf_arib_b67(float x); @@ -110,4 +110,9 @@ float ff_eotf_arib_b67(float x);
float inverse_eotf_arib_b67(float x); float ff_inverse_eotf_arib_b67(float x);
float inverse_eotf_bt1886(float x); float ff_inverse_eotf_bt1886(float x);
+int ff_get_range_off(int *off, int *y_rng, int *uv_rng, +int ff_get_range_off(int *off, int *y_rng, int *uv_rng,
+ enum AVColorRange rng, int depth); + enum AVColorRange rng, int depth);
+void ff_get_yuv_coeffs(int out[3][3][8], double (*table)[3], +void ff_get_yuv_coeffs(int out[3][3][8], double (*table)[3],
+ int depth, int y_rng, int uv_rng, int yuv2rgb); + int depth, int y_rng, int uv_rng, int yuv2rgb);
+
#endif #endif
Index: FFmpeg/libavfilter/vf_tonemapx.c Index: FFmpeg/libavfilter/vf_tonemapx.c
=================================================================== ===================================================================
--- /dev/null --- /dev/null
+++ FFmpeg/libavfilter/vf_tonemapx.c +++ FFmpeg/libavfilter/vf_tonemapx.c
@@ -0,0 +1,1948 @@ @@ -0,0 +1,1972 @@
+/* +/*
+ * This file is part of FFmpeg. + * This file is part of FFmpeg.
+ * + *
@@ -2708,7 +2710,8 @@ Index: FFmpeg/libavfilter/vf_tonemapx.c
+typedef struct TonemapxContext { +typedef struct TonemapxContext {
+ const AVClass *class; + const AVClass *class;
+ +
+ enum TonemapAlgorithm tonemap; + /* enum TonemapAlgorithm */
+ int tonemap;
+ enum AVColorTransferCharacteristic trc; + enum AVColorTransferCharacteristic trc;
+ enum AVColorSpace spc; + enum AVColorSpace spc;
+ enum AVColorPrimaries pri; + enum AVColorPrimaries pri;
@@ -2729,7 +2732,7 @@ Index: FFmpeg/libavfilter/vf_tonemapx.c
+ uint16_t *delin_lut; + uint16_t *delin_lut;
+ int in_yuv_off, out_yuv_off; + int in_yuv_off, out_yuv_off;
+ +
+ struct DoviMetadata *dovi; + struct FFDOVIMetadataRemap *dovi;
+ +
+ DECLARE_ALIGNED(16, float, dovi_pbuf)[3*(params_sz+pivots_sz+coeffs_sz+mmr_sz)]; + DECLARE_ALIGNED(16, float, dovi_pbuf)[3*(params_sz+pivots_sz+coeffs_sz+mmr_sz)];
+ DECLARE_ALIGNED(16, int, yuv2rgb_coeffs)[3][3][8]; + DECLARE_ALIGNED(16, int, yuv2rgb_coeffs)[3][3][8];
@@ -2827,7 +2830,7 @@ Index: FFmpeg/libavfilter/vf_tonemapx.c
+ for (c = 0; c < 3; c++) { + for (c = 0; c < 3; c++) {
+ int has_poly = 0, has_mmr = 0, mmr_single = 1; + int has_poly = 0, has_mmr = 0, mmr_single = 1;
+ int mmr_idx = 0, min_order = 3, max_order = 1; + int mmr_idx = 0, min_order = 3, max_order = 1;
+ const struct ReshapeData *comp = &s->dovi->comp[c]; + const struct FFDOVIReshapeData *comp = &s->dovi->comp[c];
+ if (!comp->num_pivots) + if (!comp->num_pivots)
+ continue; + continue;
+ av_assert0(comp->num_pivots >= 2 && comp->num_pivots <= 9); + av_assert0(comp->num_pivots >= 2 && comp->num_pivots <= 9);
@@ -3099,13 +3102,13 @@ Index: FFmpeg/libavfilter/vf_tonemapx.c
+ +
+static float bt2390(float s, float knee_offset, float peak) +static float bt2390(float s, float knee_offset, float peak)
+{ +{
+ float peak_pq = inverse_eotf_st2084(peak, REFERENCE_WHITE_ALT); + float peak_pq = ff_inverse_eotf_st2084(peak, REFERENCE_WHITE_ALT);
+ float scale = peak_pq > 0.0f ? (1.0f / peak_pq) : 1.0f; + float scale = peak_pq > 0.0f ? (1.0f / peak_pq) : 1.0f;
+ +
+ // SDR peak + // SDR peak
+ float dst_peak = 1.0f; + float dst_peak = 1.0f;
+ float s_pq = inverse_eotf_st2084(s, REFERENCE_WHITE_ALT) * scale; + float s_pq = ff_inverse_eotf_st2084(s, REFERENCE_WHITE_ALT) * scale;
+ float max_lum = inverse_eotf_st2084(dst_peak, REFERENCE_WHITE_ALT) * scale; + float max_lum = ff_inverse_eotf_st2084(dst_peak, REFERENCE_WHITE_ALT) * scale;
+ +
+ float ks = (1.0f + knee_offset) * max_lum - knee_offset; + float ks = (1.0f + knee_offset) * max_lum - knee_offset;
+ float tb = (s_pq - ks) / (1.0f - ks); + float tb = (s_pq - ks) / (1.0f - ks);
@@ -3116,7 +3119,7 @@ Index: FFmpeg/libavfilter/vf_tonemapx.c
+ (-2.0f * tb3 + 3.0f * tb2) * max_lum; + (-2.0f * tb3 + 3.0f * tb2) * max_lum;
+ float sig = MIX(pb, s_pq, s_pq < ks); + float sig = MIX(pb, s_pq, s_pq < ks);
+ +
+ return eotf_st2084(sig * peak_pq, REFERENCE_WHITE_ALT); + return ff_eotf_st2084(sig * peak_pq, REFERENCE_WHITE_ALT);
+} +}
+ +
+static float mapsig(enum TonemapAlgorithm alg, float sig, double peak, double param) +static float mapsig(enum TonemapAlgorithm alg, float sig, double peak, double param)
@@ -3157,9 +3160,9 @@ Index: FFmpeg/libavfilter/vf_tonemapx.c
+static float linearize(float x, enum AVColorTransferCharacteristic trc_src) +static float linearize(float x, enum AVColorTransferCharacteristic trc_src)
+{ +{
+ if (trc_src == AVCOL_TRC_SMPTE2084) + if (trc_src == AVCOL_TRC_SMPTE2084)
+ return eotf_st2084(x, REFERENCE_WHITE_ALT); + return ff_eotf_st2084(x, REFERENCE_WHITE_ALT);
+ else if (trc_src == AVCOL_TRC_ARIB_STD_B67) + else if (trc_src == AVCOL_TRC_ARIB_STD_B67)
+ return eotf_arib_b67(x); + return ff_eotf_arib_b67(x);
+ else + else
+ return x; + return x;
+} +}
@@ -3167,7 +3170,7 @@ Index: FFmpeg/libavfilter/vf_tonemapx.c
+static float delinearize(float x, enum AVColorTransferCharacteristic trc_dst) +static float delinearize(float x, enum AVColorTransferCharacteristic trc_dst)
+{ +{
+ if (trc_dst == AVCOL_TRC_BT709 || trc_dst == AVCOL_TRC_BT2020_10) + if (trc_dst == AVCOL_TRC_BT709 || trc_dst == AVCOL_TRC_BT2020_10)
+ return inverse_eotf_bt1886(x); + return ff_inverse_eotf_bt1886(x);
+ else + else
+ return x; + return x;
+} +}
@@ -4162,7 +4165,7 @@ Index: FFmpeg/libavfilter/vf_tonemapx.c
+ // such videos usually have vdr_rpu_profile == 0, for example profile 5 videos + // such videos usually have vdr_rpu_profile == 0, for example profile 5 videos
+ // this could be wrong as there is no public documentation on this field + // this could be wrong as there is no public documentation on this field
+ if (rpu->disable_residual_flag && rpu->vdr_rpu_profile == 0) { + if (rpu->disable_residual_flag && rpu->vdr_rpu_profile == 0) {
+ struct DoviMetadata *dovi = av_malloc(sizeof(*dovi)); + struct FFDOVIMetadataRemap *dovi = av_malloc(sizeof(*dovi));
+ s->dovi = dovi; + s->dovi = dovi;
+ if (!s->dovi) + if (!s->dovi)
+ goto fail; + goto fail;
@@ -4248,7 +4251,7 @@ Index: FFmpeg/libavfilter/vf_tonemapx.c
+ +
+ av_frame_free(&in); + av_frame_free(&in);
+ +
+ if (s->trc !=AVCOL_TRC_SMPTE2084) { + if (s->trc != AVCOL_TRC_SMPTE2084) {
+ av_frame_remove_side_data(out, AV_FRAME_DATA_MASTERING_DISPLAY_METADATA); + av_frame_remove_side_data(out, AV_FRAME_DATA_MASTERING_DISPLAY_METADATA);
+ av_frame_remove_side_data(out, AV_FRAME_DATA_CONTENT_LIGHT_LEVEL); + av_frame_remove_side_data(out, AV_FRAME_DATA_CONTENT_LIGHT_LEVEL);
+ } + }
@@ -4516,6 +4519,21 @@ Index: FFmpeg/libavfilter/vf_tonemapx.c
+ return 0; + return 0;
+} +}
+ +
+static int config_props(AVFilterLink *outlink)
+{
+ AVFilterContext *ctx = outlink->src;
+ TonemapxContext *s = ctx->priv;
+
+ if (s->trc != AVCOL_TRC_SMPTE2084) {
+ av_frame_side_data_remove(&outlink->side_data, &outlink->nb_side_data,
+ AV_FRAME_DATA_MASTERING_DISPLAY_METADATA);
+ av_frame_side_data_remove(&outlink->side_data, &outlink->nb_side_data,
+ AV_FRAME_DATA_CONTENT_LIGHT_LEVEL);
+ }
+
+ return 0;
+}
+
+#define OFFSET(x) offsetof(TonemapxContext, x) +#define OFFSET(x) offsetof(TonemapxContext, x)
+#define FLAGS AV_OPT_FLAG_VIDEO_PARAM | AV_OPT_FLAG_FILTERING_PARAM +#define FLAGS AV_OPT_FLAG_VIDEO_PARAM | AV_OPT_FLAG_FILTERING_PARAM
+static const AVOption tonemapx_options[] = { +static const AVOption tonemapx_options[] = {
@@ -4565,17 +4583,25 @@ Index: FFmpeg/libavfilter/vf_tonemapx.c
+ }, + },
+}; +};
+ +
+AVFilter ff_vf_tonemapx = { +static const AVFilterPad tonemapx_outputs[] = {
+ .name = "tonemapx", + {
+ .description = NULL_IF_CONFIG_SMALL("SIMD optimized HDR to SDR tonemapping"), + .name = "default",
+ .type = AVMEDIA_TYPE_VIDEO,
+ .config_props = config_props,
+ },
+};
+
+const FFFilter ff_vf_tonemapx = {
+ .p.name = "tonemapx",
+ .p.description = NULL_IF_CONFIG_SMALL("SIMD optimized HDR to SDR tonemapping"),
+ .init = init, + .init = init,
+ .uninit = uninit, + .uninit = uninit,
+ .priv_size = sizeof(TonemapxContext), + .priv_size = sizeof(TonemapxContext),
+ .priv_class = &tonemapx_class, + .p.priv_class = &tonemapx_class,
+ FILTER_INPUTS(tonemapx_inputs), + FILTER_INPUTS(tonemapx_inputs),
+ FILTER_OUTPUTS(ff_video_default_filterpad), + FILTER_OUTPUTS(tonemapx_outputs),
+ FILTER_QUERY_FUNC(query_formats), + FILTER_QUERY_FUNC(query_formats),
+ .flags = AVFILTER_FLAG_SLICE_THREADS, + .p.flags = AVFILTER_FLAG_SLICE_THREADS,
+}; +};
Index: FFmpeg/libavfilter/vf_tonemapx.h Index: FFmpeg/libavfilter/vf_tonemapx.h
=================================================================== ===================================================================
@@ -4663,7 +4689,7 @@ Index: FFmpeg/libavfilter/vf_tonemapx.h
+ int rgb2rgb_passthrough; + int rgb2rgb_passthrough;
+ const AVLumaCoefficients *coeffs, *ocoeffs; + const AVLumaCoefficients *coeffs, *ocoeffs;
+ double desat; + double desat;
+ struct DoviMetadata *dovi; + struct FFDOVIMetadataRemap *dovi;
+ float *dovi_pbuf; + float *dovi_pbuf;
+ double (*lms2rgb_matrix)[3][3]; + double (*lms2rgb_matrix)[3][3];
+ float (*ycc_offset)[3]; + float (*ycc_offset)[3];
@@ -4730,15 +4756,14 @@ Index: FFmpeg/libavfilter/x86/Makefile
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/x86/Makefile --- FFmpeg.orig/libavfilter/x86/Makefile
+++ FFmpeg/libavfilter/x86/Makefile +++ FFmpeg/libavfilter/x86/Makefile
@@ -36,6 +36,8 @@ OBJS-$(CONFIG_STEREO3D_FILTER) @@ -1,5 +1,7 @@
OBJS-$(CONFIG_TBLEND_FILTER) += x86/vf_blend_init.o OBJS-$(CONFIG_NOISE_FILTER) += x86/vf_noise.o
OBJS-$(CONFIG_THRESHOLD_FILTER) += x86/vf_threshold_init.o OBJS-$(CONFIG_SPP_FILTER) += x86/vf_spp.o
OBJS-$(CONFIG_TINTERLACE_FILTER) += x86/vf_tinterlace_init.o
+OBJS-$(CONFIG_TONEMAPX_FILTER) += x86/vf_tonemapx_intrin_sse.o \ +OBJS-$(CONFIG_TONEMAPX_FILTER) += x86/vf_tonemapx_intrin_sse.o \
+ x86/vf_tonemapx_intrin_avx.o + x86/vf_tonemapx_intrin_avx.o
OBJS-$(CONFIG_TRANSPOSE_FILTER) += x86/vf_transpose_init.o
OBJS-$(CONFIG_VOLUME_FILTER) += x86/af_volume_init.o X86ASM-OBJS-$(CONFIG_SCENE_SAD) += x86/scene_sad.o x86/scene_sad_init.o
OBJS-$(CONFIG_V360_FILTER) += x86/vf_v360_init.o
Index: FFmpeg/libavfilter/x86/vf_tonemapx_intrin_avx.c Index: FFmpeg/libavfilter/x86/vf_tonemapx_intrin_avx.c
=================================================================== ===================================================================
--- /dev/null --- /dev/null
@@ -7405,7 +7430,7 @@ Index: FFmpeg/libavfilter/x86/vf_tonemapx_intrin_sse.c
=================================================================== ===================================================================
--- /dev/null --- /dev/null
+++ FFmpeg/libavfilter/x86/vf_tonemapx_intrin_sse.c +++ FFmpeg/libavfilter/x86/vf_tonemapx_intrin_sse.c
@@ -0,0 +1,2740 @@ @@ -0,0 +1,2742 @@
+/* +/*
+ * Copyright (c) 2024 Gnattu OC <gnattuoc@me.com> + * Copyright (c) 2024 Gnattu OC <gnattuoc@me.com>
+ * + *
@@ -7470,6 +7495,7 @@ Index: FFmpeg/libavfilter/x86/vf_tonemapx_intrin_sse.c
+ return _mm_or_si128(_mm_and_si128(cmp, a), _mm_andnot_si128(cmp, xor_result)); + return _mm_or_si128(_mm_and_si128(cmp, a), _mm_andnot_si128(cmp, xor_result));
+} +}
+ +
+/*
+X86_64_V2 inline static __m128 mix_float32x4(__m128 x, __m128 y, __m128 a) +X86_64_V2 inline static __m128 mix_float32x4(__m128 x, __m128 y, __m128 a)
+{ +{
+ __m128 n = _mm_sub_ps(y, x); + __m128 n = _mm_sub_ps(y, x);
@@ -7477,6 +7503,7 @@ Index: FFmpeg/libavfilter/x86/vf_tonemapx_intrin_sse.c
+ n = _mm_add_ps(n, x); + n = _mm_add_ps(n, x);
+ return n; + return n;
+} +}
+*/
+ +
+X86_64_V2 inline static float reduce_floatx4(__m128 x) { +X86_64_V2 inline static float reduce_floatx4(__m128 x) {
+ x = _mm_hadd_ps(x, x); + x = _mm_hadd_ps(x, x);
@@ -1,38 +1,60 @@
Index: FFmpeg/configure
===================================================================
--- FFmpeg.orig/configure
+++ FFmpeg/configure
@@ -3076,6 +3076,8 @@ ac3_decoder_select="ac3_parser ac3dsp bs
ac3_fixed_decoder_select="ac3_parser ac3dsp bswapdsp"
ac3_encoder_select="ac3dsp audiodsp me_cmp"
ac3_fixed_encoder_select="ac3dsp audiodsp me_cmp"
+ac4_decoder_select="ac4_parser"
+ac4_decoder_deps="gplv3"
acelp_kelvin_decoder_select="audiodsp celp_math"
adpcm_circus_decoder_deps="lgpl_gpl"
adpcm_g722_decoder_select="g722dsp"
@@ -3698,6 +3700,7 @@ vvc_qsv_decoder_select="vvc_mp4toannexb_
# parsers
aac_parser_select="adts_header mpeg4audio"
+ac4_parser_deps="gplv3"
ahx_parser_deps="lgpl_gpl"
apv_parser_select="cbs_apv"
av1_parser_select="cbs_av1"
Index: FFmpeg/libavcodec/Makefile Index: FFmpeg/libavcodec/Makefile
=================================================================== ===================================================================
--- FFmpeg.orig/libavcodec/Makefile --- FFmpeg.orig/libavcodec/Makefile
+++ FFmpeg/libavcodec/Makefile +++ FFmpeg/libavcodec/Makefile
@@ -64,6 +64,8 @@ OBJS = ac3_parser.o @@ -65,6 +65,7 @@ OBJS = ac3_parser.o
vorbis_parser.o \
xiph.o \ xiph.o \
+SHLIBOBJS = float_fmul_reverse.o \ SHLIBOBJS = float_scalarproduct.o \
+ + float_fmul_reverse.o \
timecode_internal.o \
# subsystems # subsystems
include $(SRC_PATH)/libavcodec/aac/Makefile @@ -219,6 +220,7 @@ OBJS-$(CONFIG_AC3_ENCODER) +
include $(SRC_PATH)/libavcodec/hevc/Makefile
@@ -210,6 +212,7 @@ OBJS-$(CONFIG_AC3_ENCODER) +
ac3.o kbdwin.o ac3.o kbdwin.o
OBJS-$(CONFIG_AC3_FIXED_ENCODER) += ac3enc_fixed.o ac3enc.o ac3tab.o ac3.o kbdwin.o OBJS-$(CONFIG_AC3_FIXED_ENCODER) += ac3enc_fixed.o ac3enc.o ac3tab.o ac3.o kbdwin.o
OBJS-$(CONFIG_AC3_MF_ENCODER) += mfenc.o mf_utils.o OBJS-$(CONFIG_AC3_MF_ENCODER) += mfenc.o mf_utils.o
+OBJS-$(CONFIG_AC4_DECODER) += ac4dec.o kbdwin.o +OBJS-$(CONFIG_AC4_DECODER) += ac4dec.o kbdwin.o
OBJS-$(CONFIG_ACELP_KELVIN_DECODER) += g729dec.o lsp.o celp_math.o celp_filters.o acelp_filters.o acelp_pitch_delay.o acelp_vectors.o g729postfilter.o OBJS-$(CONFIG_ACELP_KELVIN_DECODER) += g729dec.o lsp.o celp_filters.o acelp_filters.o acelp_pitch_delay.o acelp_vectors.o g729postfilter.o
OBJS-$(CONFIG_AGM_DECODER) += agm.o jpegquanttables.o OBJS-$(CONFIG_AGM_DECODER) += agm.o jpegquanttables.o
OBJS-$(CONFIG_AIC_DECODER) += aic.o OBJS-$(CONFIG_AHX_DECODER) += mpegaudiodec_float.o
@@ -1189,6 +1192,7 @@ OBJS-$(CONFIG_AAC_LATM_PARSER) + @@ -1247,6 +1249,7 @@ OBJS-$(CONFIG_AAC_LATM_PARSER) +
OBJS-$(CONFIG_AAC_PARSER) += aac_parser.o aac_ac3_parser.o OBJS-$(CONFIG_AAC_PARSER) += aac_parser.o aac_ac3_parser.o
OBJS-$(CONFIG_AC3_PARSER) += aac_ac3_parser.o ac3tab.o \ OBJS-$(CONFIG_AC3_PARSER) += aac_ac3_parser.o ac3tab.o \
ac3_channel_layout_tab.o ac3_channel_layout_tab.o
+OBJS-$(CONFIG_AC4_PARSER) += ac4_parser.o +OBJS-$(CONFIG_AC4_PARSER) += ac4_parser.o
OBJS-$(CONFIG_ADX_PARSER) += adx_parser.o OBJS-$(CONFIG_ADX_PARSER) += adx_parser.o
OBJS-$(CONFIG_AHX_PARSER) += ahx_parser.o
OBJS-$(CONFIG_AMR_PARSER) += amr_parser.o OBJS-$(CONFIG_AMR_PARSER) += amr_parser.o
OBJS-$(CONFIG_AV1_PARSER) += av1_parser.o av1_parse.o
Index: FFmpeg/libavcodec/ac4_parser.c Index: FFmpeg/libavcodec/ac4_parser.c
=================================================================== ===================================================================
--- /dev/null --- /dev/null
+++ FFmpeg/libavcodec/ac4_parser.c +++ FFmpeg/libavcodec/ac4_parser.c
@@ -0,0 +1,96 @@ @@ -0,0 +1,99 @@
+/* +/*
+ * This file is placed in the public domain.
+ *
+ * Copyright (c) 2025 Paul B Mahol + * Copyright (c) 2025 Paul B Mahol
+ * + *
+ * This file is part of Librempeg + * This file is part of Librempeg
@@ -64,6 +86,7 @@ Index: FFmpeg/libavcodec/ac4_parser.c
+#include "ac4dec_data.h" +#include "ac4dec_data.h"
+#include "get_bits.h" +#include "get_bits.h"
+#include "parser.h" +#include "parser.h"
+#include "parser_internal.h"
+ +
+static int variable_bits(GetBitContext *gb, int bits) +static int variable_bits(GetBitContext *gb, int bits)
+{ +{
@@ -124,9 +147,9 @@ Index: FFmpeg/libavcodec/ac4_parser.c
+ return buf_size; + return buf_size;
+} +}
+ +
+const AVCodecParser ff_ac4_parser = { +const FFCodecParser ff_ac4_parser = {
+ .codec_ids = { AV_CODEC_ID_AC4 }, + PARSER_CODEC_LIST(AV_CODEC_ID_AC4),
+ .parser_parse = ac4_parse, + .parse = ac4_parse,
+}; +};
Index: FFmpeg/libavcodec/ac4dec.c Index: FFmpeg/libavcodec/ac4dec.c
=================================================================== ===================================================================
@@ -134,6 +157,8 @@ Index: FFmpeg/libavcodec/ac4dec.c
+++ FFmpeg/libavcodec/ac4dec.c +++ FFmpeg/libavcodec/ac4dec.c
@@ -0,0 +1,6763 @@ @@ -0,0 +1,6763 @@
+/* +/*
+ * This file is placed in the public domain.
+ *
+ * AC-4 Audio Decoder + * AC-4 Audio Decoder
+ * + *
+ * Copyright (c) 2019 Paul B Mahol + * Copyright (c) 2019 Paul B Mahol
@@ -4533,8 +4558,8 @@ Index: FFmpeg/libavcodec/ac4dec.c
+ return ret; + return ret;
+ } + }
+ +
+ if (ss->im_codec_mode == IM_ASPX_AJCC) + // if (ss->im_codec_mode == IM_ASPX_AJCC)
+ ; + // ;
+ +
+ if (ss->im_codec_mode == IM_SCPL || + if (ss->im_codec_mode == IM_SCPL ||
+ ss->im_codec_mode == IM_ASPX_SCPL || + ss->im_codec_mode == IM_ASPX_SCPL ||
@@ -6892,16 +6917,16 @@ Index: FFmpeg/libavcodec/ac4dec.c
+ FF_CODEC_DECODE_CB(ac4_decode_frame), + FF_CODEC_DECODE_CB(ac4_decode_frame),
+ .flush = ac4_flush, + .flush = ac4_flush,
+ .p.capabilities = AV_CODEC_CAP_DR1 | AV_CODEC_CAP_CHANNEL_CONF, + .p.capabilities = AV_CODEC_CAP_DR1 | AV_CODEC_CAP_CHANNEL_CONF,
+ .p.sample_fmts = (const enum AVSampleFormat[]) { AV_SAMPLE_FMT_FLTP,
+ AV_SAMPLE_FMT_NONE },
+ .caps_internal = FF_CODEC_CAP_INIT_CLEANUP, + .caps_internal = FF_CODEC_CAP_INIT_CLEANUP,
+}; +};
Index: FFmpeg/libavcodec/ac4dec_data.h Index: FFmpeg/libavcodec/ac4dec_data.h
=================================================================== ===================================================================
--- /dev/null --- /dev/null
+++ FFmpeg/libavcodec/ac4dec_data.h +++ FFmpeg/libavcodec/ac4dec_data.h
@@ -0,0 +1,1688 @@ @@ -0,0 +1,1690 @@
+/* +/*
+ * This file is placed in the public domain.
+ *
+ * AC-4 Audio Decoder + * AC-4 Audio Decoder
+ * + *
+ * This file is part of Librempeg + * This file is part of Librempeg
@@ -8593,20 +8618,22 @@ Index: FFmpeg/libavcodec/allcodecs.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavcodec/allcodecs.c --- FFmpeg.orig/libavcodec/allcodecs.c
+++ FFmpeg/libavcodec/allcodecs.c +++ FFmpeg/libavcodec/allcodecs.c
@@ -430,6 +430,7 @@ extern const FFCodec ff_ac3_encoder; @@ -442,6 +442,7 @@ extern const FFCodec ff_ac3_encoder;
extern const FFCodec ff_ac3_decoder; extern const FFCodec ff_ac3_decoder;
extern const FFCodec ff_ac3_fixed_encoder; extern const FFCodec ff_ac3_fixed_encoder;
extern const FFCodec ff_ac3_fixed_decoder; extern const FFCodec ff_ac3_fixed_decoder;
+extern const FFCodec ff_ac4_decoder; +extern const FFCodec ff_ac4_decoder;
extern const FFCodec ff_acelp_kelvin_decoder; extern const FFCodec ff_acelp_kelvin_decoder;
extern const FFCodec ff_ahx_decoder;
extern const FFCodec ff_alac_encoder; extern const FFCodec ff_alac_encoder;
extern const FFCodec ff_alac_decoder;
Index: FFmpeg/libavcodec/float_fmul_reverse.c Index: FFmpeg/libavcodec/float_fmul_reverse.c
=================================================================== ===================================================================
--- /dev/null --- /dev/null
+++ FFmpeg/libavcodec/float_fmul_reverse.c +++ FFmpeg/libavcodec/float_fmul_reverse.c
@@ -0,0 +1,19 @@ @@ -0,0 +1,21 @@
+/* +/*
+ * This file is placed in the public domain.
+ *
+ * This file is part of Librempeg + * This file is part of Librempeg
+ * + *
+ * Librempeg is free software; you can redistribute it and/or modify + * Librempeg is free software; you can redistribute it and/or modify
@@ -8642,26 +8669,26 @@ Index: FFmpeg/libavcodec/parsers.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavcodec/parsers.c --- FFmpeg.orig/libavcodec/parsers.c
+++ FFmpeg/libavcodec/parsers.c +++ FFmpeg/libavcodec/parsers.c
@@ -23,6 +23,7 @@ @@ -40,6 +40,7 @@ FF_ENABLE_DEPRECATION_WARNINGS
extern const AVCodecParser ff_aac_parser; extern const FFCodecParser ff_aac_parser;
extern const AVCodecParser ff_aac_latm_parser; extern const FFCodecParser ff_aac_latm_parser;
extern const AVCodecParser ff_ac3_parser; extern const FFCodecParser ff_ac3_parser;
+extern const AVCodecParser ff_ac4_parser; +extern const AVCodecParser ff_ac4_parser;
extern const AVCodecParser ff_adx_parser; extern const FFCodecParser ff_adx_parser;
extern const AVCodecParser ff_amr_parser; extern const FFCodecParser ff_ahx_parser;
extern const AVCodecParser ff_av1_parser; extern const FFCodecParser ff_amr_parser;
Index: FFmpeg/libavutil/Makefile Index: FFmpeg/libavutil/Makefile
=================================================================== ===================================================================
--- FFmpeg.orig/libavutil/Makefile --- FFmpeg.orig/libavutil/Makefile
+++ FFmpeg/libavutil/Makefile +++ FFmpeg/libavutil/Makefile
@@ -136,6 +136,7 @@ OBJS = adler32.o @@ -142,6 +142,7 @@ OBJS = adler32.o
file.o \ file.o \
file_open.o \ file_open.o \
float_dsp.o \ float_dsp.o \
+ float_fmul_reverse.o \ + float_fmul_reverse.o \
float_scalarproduct.o \
film_grain_params.o \
fixed_dsp.o \ fixed_dsp.o \
frame.o \
hash.o \
Index: FFmpeg/libavutil/float_dsp.c Index: FFmpeg/libavutil/float_dsp.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavutil/float_dsp.c --- FFmpeg.orig/libavutil/float_dsp.c
@@ -8683,20 +8710,20 @@ Index: FFmpeg/libavutil/float_dsp.c
static void butterflies_float_c(float *restrict v1, float *restrict v2, static void butterflies_float_c(float *restrict v1, float *restrict v2,
int len) int len)
{ {
@@ -157,7 +147,7 @@ av_cold AVFloatDSPContext *avpriv_float_ @@ -146,7 +136,7 @@ av_cold AVFloatDSPContext *avpriv_float_
fdsp->vector_dmul_scalar = vector_dmul_scalar_c; fdsp->vector_dmul_scalar = vector_dmul_scalar_c;
fdsp->vector_fmul_window = vector_fmul_window_c; fdsp->vector_fmul_window = vector_fmul_window_c;
fdsp->vector_fmul_add = vector_fmul_add_c; fdsp->vector_fmul_add = vector_fmul_add_c;
- fdsp->vector_fmul_reverse = vector_fmul_reverse_c; - fdsp->vector_fmul_reverse = vector_fmul_reverse_c;
+ fdsp->vector_fmul_reverse = ff_vector_fmul_reverse_c; + fdsp->vector_fmul_reverse = ff_vector_fmul_reverse_c;
fdsp->butterflies_float = butterflies_float_c; fdsp->butterflies_float = butterflies_float_c;
fdsp->scalarproduct_float = avpriv_scalarproduct_float_c; fdsp->scalarproduct_float = ff_scalarproduct_float_c;
fdsp->scalarproduct_double = ff_scalarproduct_double_c; fdsp->scalarproduct_double = ff_scalarproduct_double_c;
Index: FFmpeg/libavutil/float_dsp.h Index: FFmpeg/libavutil/float_dsp.h
=================================================================== ===================================================================
--- FFmpeg.orig/libavutil/float_dsp.h --- FFmpeg.orig/libavutil/float_dsp.h
+++ FFmpeg/libavutil/float_dsp.h +++ FFmpeg/libavutil/float_dsp.h
@@ -229,6 +229,19 @@ float avpriv_scalarproduct_float_c(const @@ -229,6 +229,19 @@ float ff_scalarproduct_float_c(const flo
double ff_scalarproduct_double_c(const double *v1, const double *v2, double ff_scalarproduct_double_c(const double *v1, const double *v2,
size_t len); size_t len);
@@ -8720,8 +8747,10 @@ Index: FFmpeg/libavutil/float_fmul_reverse.c
=================================================================== ===================================================================
--- /dev/null --- /dev/null
+++ FFmpeg/libavutil/float_fmul_reverse.c +++ FFmpeg/libavutil/float_fmul_reverse.c
@@ -0,0 +1,29 @@ @@ -0,0 +1,31 @@
+/* +/*
+ * This file is placed in the public domain.
+ *
+ * This file is part of Librempeg + * This file is part of Librempeg
+ * + *
+ * Librempeg is free software; you can redistribute it and/or modify + * Librempeg is free software; you can redistribute it and/or modify
@@ -2,19 +2,19 @@ Index: FFmpeg/configure
=================================================================== ===================================================================
--- FFmpeg.orig/configure --- FFmpeg.orig/configure
+++ FFmpeg/configure +++ FFmpeg/configure
@@ -3863,6 +3863,7 @@ boxblur_opencl_filter_deps="opencl gpl" @@ -4120,6 +4120,7 @@ boxblur_opencl_filter_deps="opencl gpl"
bs2b_filter_deps="libbs2b" bs2b_filter_deps="libbs2b"
bwdif_cuda_filter_deps="ffnvcodec" bwdif_cuda_filter_deps="ffnvcodec"
bwdif_cuda_filter_deps_any="cuda_nvcc cuda_llvm" bwdif_cuda_filter_deps_any="cuda_nvcc cuda_llvm"
+bwdif_videotoolbox_filter_deps="metal corevideo videotoolbox" +bwdif_videotoolbox_filter_deps="metal corevideo videotoolbox"
bwdif_vulkan_filter_deps="vulkan spirv_compiler" bwdif_vulkan_filter_deps="vulkan spirv_compiler"
chromaber_vulkan_filter_deps="vulkan spirv_compiler" chromaber_vulkan_filter_deps="vulkan spirv_library"
color_vulkan_filter_deps="vulkan spirv_compiler" color_vulkan_filter_deps="vulkan spirv_library"
Index: FFmpeg/libavfilter/Makefile Index: FFmpeg/libavfilter/Makefile
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/Makefile --- FFmpeg.orig/libavfilter/Makefile
+++ FFmpeg/libavfilter/Makefile +++ FFmpeg/libavfilter/Makefile
@@ -219,6 +219,10 @@ OBJS-$(CONFIG_BOXBLUR_OPENCL_FILTER) @@ -224,6 +224,10 @@ OBJS-$(CONFIG_BOXBLUR_OPENCL_FILTER)
OBJS-$(CONFIG_BWDIF_FILTER) += vf_bwdif.o bwdifdsp.o yadif_common.o OBJS-$(CONFIG_BWDIF_FILTER) += vf_bwdif.o bwdifdsp.o yadif_common.o
OBJS-$(CONFIG_BWDIF_CUDA_FILTER) += vf_bwdif_cuda.o vf_bwdif_cuda.ptx.o \ OBJS-$(CONFIG_BWDIF_CUDA_FILTER) += vf_bwdif_cuda.o vf_bwdif_cuda.ptx.o \
yadif_common.o yadif_common.o
@@ -29,19 +29,19 @@ Index: FFmpeg/libavfilter/allfilters.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/allfilters.c --- FFmpeg.orig/libavfilter/allfilters.c
+++ FFmpeg/libavfilter/allfilters.c +++ FFmpeg/libavfilter/allfilters.c
@@ -201,6 +201,7 @@ extern const AVFilter ff_vf_boxblur; @@ -204,6 +204,7 @@ extern const FFFilter ff_vf_boxblur;
extern const AVFilter ff_vf_boxblur_opencl; extern const FFFilter ff_vf_boxblur_opencl;
extern const AVFilter ff_vf_bwdif; extern const FFFilter ff_vf_bwdif;
extern const AVFilter ff_vf_bwdif_cuda; extern const FFFilter ff_vf_bwdif_cuda;
+extern const AVFilter ff_vf_bwdif_videotoolbox; +extern const FFFilter ff_vf_bwdif_videotoolbox;
extern const AVFilter ff_vf_bwdif_vulkan; extern const FFFilter ff_vf_bwdif_vulkan;
extern const AVFilter ff_vf_cas; extern const FFFilter ff_vf_cas;
extern const AVFilter ff_vf_ccrepack; extern const FFFilter ff_vf_ccrepack;
Index: FFmpeg/libavfilter/metal/vf_bwdif_videotoolbox.metal Index: FFmpeg/libavfilter/metal/vf_bwdif_videotoolbox.metal
=================================================================== ===================================================================
--- /dev/null --- /dev/null
+++ FFmpeg/libavfilter/metal/vf_bwdif_videotoolbox.metal +++ FFmpeg/libavfilter/metal/vf_bwdif_videotoolbox.metal
@@ -0,0 +1,271 @@ @@ -0,0 +1,270 @@
+/* bwdif.metal +/* bwdif.metal
+ +
+ Copyright (c) 2003-2024 HandBrake Team + Copyright (c) 2003-2024 HandBrake Team
@@ -312,12 +312,11 @@ Index: FFmpeg/libavfilter/metal/vf_bwdif_videotoolbox.metal
+ } + }
+ dst.write(pred.xyyy, pos); + dst.write(pred.xyyy, pos);
+} +}
+
Index: FFmpeg/libavfilter/vf_bwdif_videotoolbox.m Index: FFmpeg/libavfilter/vf_bwdif_videotoolbox.m
=================================================================== ===================================================================
--- /dev/null --- /dev/null
+++ FFmpeg/libavfilter/vf_bwdif_videotoolbox.m +++ FFmpeg/libavfilter/vf_bwdif_videotoolbox.m
@@ -0,0 +1,448 @@ @@ -0,0 +1,449 @@
+/* +/*
+ * Copyright (C) 2018 Philip Langdale <philipl@overt.org> + * Copyright (C) 2018 Philip Langdale <philipl@overt.org>
+ * 2020 Aman Karmani <aman@tmm1.net> + * 2020 Aman Karmani <aman@tmm1.net>
@@ -753,16 +752,31 @@ Index: FFmpeg/libavfilter/vf_bwdif_videotoolbox.m
+ }, + },
+}; +};
+ +
+const AVFilter ff_vf_bwdif_videotoolbox = { +const FFFilter ff_vf_bwdif_videotoolbox = {
+ .name = "bwdif_videotoolbox", + .p.name = "bwdif_videotoolbox",
+ .description = NULL_IF_CONFIG_SMALL("BWDIF for VideoToolbox frames using Metal compute"), + .p.description = NULL_IF_CONFIG_SMALL("BWDIF for VideoToolbox frames using Metal compute"),
+ .priv_size = BWDIF_VT_CTX_SIZE, + .priv_size = BWDIF_VT_CTX_SIZE,
+ .priv_class = &bwdif_videotoolbox_class, + .p.priv_class = &bwdif_videotoolbox_class,
+ .init = bwdif_videotoolbox_init, + .init = bwdif_videotoolbox_init,
+ .uninit = bwdif_videotoolbox_uninit, + .uninit = bwdif_videotoolbox_uninit,
+ FILTER_SINGLE_PIXFMT(AV_PIX_FMT_VIDEOTOOLBOX), + FILTER_SINGLE_PIXFMT(AV_PIX_FMT_VIDEOTOOLBOX),
+ FILTER_INPUTS(bwdif_videotoolbox_inputs), + FILTER_INPUTS(bwdif_videotoolbox_inputs),
+ FILTER_OUTPUTS(bwdif_videotoolbox_outputs), + FILTER_OUTPUTS(bwdif_videotoolbox_outputs),
+ .flags = AVFILTER_FLAG_SUPPORT_TIMELINE_INTERNAL, + .p.flags = AVFILTER_FLAG_HWDEVICE |
+ AVFILTER_FLAG_SUPPORT_TIMELINE_INTERNAL,
+ .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE, + .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE,
+}; +};
Index: FFmpeg/libavfilter/vf_yadif_videotoolbox.m
===================================================================
--- FFmpeg.orig/libavfilter/vf_yadif_videotoolbox.m
+++ FFmpeg/libavfilter/vf_yadif_videotoolbox.m
@@ -436,7 +436,8 @@ const FFFilter ff_vf_yadif_videotoolbox
.p.name = "yadif_videotoolbox",
.p.description = NULL_IF_CONFIG_SMALL("YADIF for VideoToolbox frames using Metal compute"),
.p.priv_class = &yadif_videotoolbox_class,
- .p.flags = AVFILTER_FLAG_SUPPORT_TIMELINE_INTERNAL,
+ .p.flags = AVFILTER_FLAG_HWDEVICE |
+ AVFILTER_FLAG_SUPPORT_TIMELINE_INTERNAL,
.priv_size = YADIF_VT_CTX_SIZE,
.init = yadif_videotoolbox_init,
.uninit = yadif_videotoolbox_uninit,
@@ -2,7 +2,7 @@ Index: FFmpeg/libavcodec/hevc/hevcdec.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavcodec/hevc/hevcdec.c --- FFmpeg.orig/libavcodec/hevc/hevcdec.c
+++ FFmpeg/libavcodec/hevc/hevcdec.c +++ FFmpeg/libavcodec/hevc/hevcdec.c
@@ -660,6 +660,9 @@ static enum AVPixelFormat get_format(HEV @@ -719,6 +719,9 @@ static enum AVPixelFormat get_format(HEV
#if CONFIG_HEVC_NVDEC_HWACCEL #if CONFIG_HEVC_NVDEC_HWACCEL
*fmt++ = AV_PIX_FMT_CUDA; *fmt++ = AV_PIX_FMT_CUDA;
#endif #endif
@@ -12,9 +12,9 @@ Index: FFmpeg/libavcodec/hevc/hevcdec.c
break; break;
case AV_PIX_FMT_YUV422P12: case AV_PIX_FMT_YUV422P12:
#if CONFIG_HEVC_DXVA2_HWACCEL #if CONFIG_HEVC_DXVA2_HWACCEL
@@ -675,6 +678,9 @@ static enum AVPixelFormat get_format(HEV @@ -737,6 +740,9 @@ static enum AVPixelFormat get_format(HEV
#if CONFIG_HEVC_VULKAN_HWACCEL #if CONFIG_HEVC_NVDEC_HWACCEL
*fmt++ = AV_PIX_FMT_VULKAN; *fmt++ = AV_PIX_FMT_CUDA;
#endif #endif
+#if CONFIG_HEVC_VIDEOTOOLBOX_HWACCEL +#if CONFIG_HEVC_VIDEOTOOLBOX_HWACCEL
+ *fmt++ = AV_PIX_FMT_VIDEOTOOLBOX; + *fmt++ = AV_PIX_FMT_VIDEOTOOLBOX;
@@ -2,13 +2,13 @@ Index: FFmpeg/fftools/ffmpeg_filter.c
=================================================================== ===================================================================
--- FFmpeg.orig/fftools/ffmpeg_filter.c --- FFmpeg.orig/fftools/ffmpeg_filter.c
+++ FFmpeg/fftools/ffmpeg_filter.c +++ FFmpeg/fftools/ffmpeg_filter.c
@@ -2651,7 +2651,8 @@ static void sub2video_heartbeat(InputFil @@ -2948,7 +2948,8 @@ static void sub2video_heartbeat(InputFil
or if we need to initialize the system, update the or if we need to initialize the system, update the
overlayed subpicture and its start/end times */ overlaid subpicture and its start/end times */
sub2video_update(ifp, pts2 + 1, NULL); sub2video_update(ifp, pts2 + 1, NULL);
- else - else
+ +
+ if (av_buffersrc_get_nb_failed_requests(ifp->filter)) + if (av_buffersrc_get_nb_failed_requests(ifp->ifilter.filter))
sub2video_push_ref(ifp, pts2); sub2video_push_ref(ifp, pts2);
} }
@@ -31,17 +31,3 @@ Index: FFmpeg/tests/ref/fate/filter-overlay-dvdsub-2397
1, 3952, 3952, 10, 2013, 0x0e5db67e 1, 3952, 3952, 10, 2013, 0x0e5db67e
1, 3962, 3962, 10, 2013, 0xfc9baf97 1, 3962, 3962, 10, 2013, 0xfc9baf97
0, 119, 119, 1, 518400, 0x588534fc 0, 119, 119, 1, 518400, 0x588534fc
Index: FFmpeg/tests/ref/fate/sub2video
===================================================================
--- FFmpeg.orig/tests/ref/fate/sub2video
+++ FFmpeg/tests/ref/fate/sub2video
@@ -68,8 +68,7 @@
0, 258, 258, 1, 518400, 0x34cdddee
0, 269, 269, 1, 518400, 0xbab197ea
1, 53910000, 53910000, 2696000, 2095, 0x61bb15ed
-0, 270, 270, 1, 518400, 0xbab197ea
-0, 271, 271, 1, 518400, 0x4db4ce51
+0, 270, 270, 1, 518400, 0x4db4ce51
0, 283, 283, 1, 518400, 0xbab197ea
1, 56663000, 56663000, 1262000, 1013, 0xc9ae89b7
0, 284, 284, 1, 518400, 0xe6bc0ea9
@@ -2,7 +2,7 @@ Index: FFmpeg/configure
=================================================================== ===================================================================
--- FFmpeg.orig/configure --- FFmpeg.orig/configure
+++ FFmpeg/configure +++ FFmpeg/configure
@@ -2533,6 +2533,7 @@ HAVE_LIST=" @@ -2687,6 +2687,7 @@ HAVE_LIST="
xmllint xmllint
zlib_gzip zlib_gzip
openvino2 openvino2
@@ -10,9 +10,9 @@ Index: FFmpeg/configure
" "
# options emitted with CONFIG_ prefix but not available on the command line # options emitted with CONFIG_ prefix but not available on the command line
@@ -6989,6 +6990,9 @@ elif enabled libvpl; then @@ -7369,6 +7370,9 @@ elif enabled libvpl; then
die "ERROR: libvpl >= 2.6 not found" die "ERROR: libvpl >= 2.6 not found"
add_cflags -DMFX_DEPRECATED_OFF add_cppflags -DMFX_DEPRECATED_OFF
check_type "vpl/mfxdefs.h vpl/mfxvideo.h" "struct mfxConfigInterface" check_type "vpl/mfxdefs.h vpl/mfxvideo.h" "struct mfxConfigInterface"
+# Ancient Intel driver on Windows still require the legacy init functions. +# Ancient Intel driver on Windows still require the legacy init functions.
+# No compiler option is changed if the test fails, just needs to add back the disabled libmfx. +# No compiler option is changed if the test fails, just needs to add back the disabled libmfx.
@@ -84,7 +84,7 @@ Index: FFmpeg/libavcodec/qsvenc.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavcodec/qsvenc.c --- FFmpeg.orig/libavcodec/qsvenc.c
+++ FFmpeg/libavcodec/qsvenc.c +++ FFmpeg/libavcodec/qsvenc.c
@@ -1192,7 +1192,7 @@ static int init_video_param(AVCodecConte @@ -1190,7 +1190,7 @@ static int init_video_param(AVCodecConte
#if QSV_HAVE_EXT_AV1_PARAM #if QSV_HAVE_EXT_AV1_PARAM
if (avctx->codec_id == AV_CODEC_ID_AV1) { if (avctx->codec_id == AV_CODEC_ID_AV1) {
@@ -93,7 +93,7 @@ Index: FFmpeg/libavcodec/qsvenc.c
q->extav1tileparam.Header.BufferId = MFX_EXTBUFF_AV1_TILE_PARAM; q->extav1tileparam.Header.BufferId = MFX_EXTBUFF_AV1_TILE_PARAM;
q->extav1tileparam.Header.BufferSz = sizeof(q->extav1tileparam); q->extav1tileparam.Header.BufferSz = sizeof(q->extav1tileparam);
q->extav1tileparam.NumTileColumns = q->tile_cols; q->extav1tileparam.NumTileColumns = q->tile_cols;
@@ -1416,7 +1416,7 @@ static int qsv_retrieve_enc_av1_params(A @@ -1414,7 +1414,7 @@ static int qsv_retrieve_enc_av1_params(A
(mfxExtBuffer*)&co3, (mfxExtBuffer*)&co3,
}; };
@@ -106,7 +106,7 @@ Index: FFmpeg/libavfilter/qsvvpp.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/qsvvpp.c --- FFmpeg.orig/libavfilter/qsvvpp.c
+++ FFmpeg/libavfilter/qsvvpp.c +++ FFmpeg/libavfilter/qsvvpp.c
@@ -1065,6 +1065,16 @@ int ff_qsvvpp_filter_frame(QSVVPPContext @@ -1066,6 +1066,16 @@ int ff_qsvvpp_filter_frame(QSVVPPContext
return 0; return 0;
} }
@@ -123,7 +123,7 @@ Index: FFmpeg/libavfilter/qsvvpp.c
#if QSV_ONEVPL #if QSV_ONEVPL
int ff_qsvvpp_create_mfx_session(void *ctx, int ff_qsvvpp_create_mfx_session(void *ctx,
@@ -1108,6 +1118,15 @@ int ff_qsvvpp_create_mfx_session(void *c @@ -1109,6 +1119,15 @@ int ff_qsvvpp_create_mfx_session(void *c
impl_idx++; impl_idx++;
} }
@@ -139,7 +139,7 @@ Index: FFmpeg/libavfilter/qsvvpp.c
if (sts < 0) if (sts < 0)
return ff_qsvvpp_print_error(ctx, sts, return ff_qsvvpp_print_error(ctx, sts,
"Error creating a MFX session"); "Error creating a MFX session");
@@ -1125,6 +1144,19 @@ int ff_qsvvpp_create_mfx_session(void *c @@ -1126,6 +1145,19 @@ int ff_qsvvpp_create_mfx_session(void *c
mfxVersion *pver, mfxVersion *pver,
mfxSession *psession) mfxSession *psession)
{ {
@@ -1,29 +0,0 @@
Index: FFmpeg/fftools/ffmpeg_sched.h
===================================================================
--- FFmpeg.orig/fftools/ffmpeg_sched.h
+++ FFmpeg/fftools/ffmpeg_sched.h
@@ -257,7 +257,10 @@ int sch_add_mux(Scheduler *sch, SchThrea
/**
* Default size of a frame thread queue.
*/
-#define DEFAULT_FRAME_THREAD_QUEUE_SIZE 8
+#define DEFAULT_FRAME_THREAD_QUEUE_SIZE 1
+
+// The new default value of 8 does little to help hwaccel, but instead increases
+// extra_hw_frames, which causes video memory on dGPU to be exhausted more easily.
/**
* Add a muxed stream for a previously added muxer.
Index: FFmpeg/tests/ref/fate/ffmpeg-fix_sub_duration_heartbeat
===================================================================
--- FFmpeg.orig/tests/ref/fate/ffmpeg-fix_sub_duration_heartbeat
+++ FFmpeg/tests/ref/fate/ffmpeg-fix_sub_duration_heartbeat
@@ -33,3 +33,8 @@
<font face="Monospace">{\an7}(<i> inaudible radio chatter</i> )
>> Safety remains our numb</font>
+9
+00:00:03,704 --> 00:00:04,004
+<font face="Monospace">{\an7}(<i> inaudible radio chatter</i> )
+>> Safety remains our number one</font>
+
@@ -1,185 +0,0 @@
Index: FFmpeg/libavutil/hwcontext_vulkan.c
===================================================================
--- FFmpeg.orig/libavutil/hwcontext_vulkan.c
+++ FFmpeg/libavutil/hwcontext_vulkan.c
@@ -90,6 +90,9 @@ typedef struct VulkanDevicePriv {
VkPhysicalDeviceMemoryProperties mprops;
VkPhysicalDeviceExternalMemoryHostPropertiesEXT hprops;
+ /* Opaque FD external semaphore properties */
+ VkExternalSemaphoreProperties ext_sem_props_opaque;
+
/* Features */
VkPhysicalDeviceVulkan11Features device_features_1_1;
VkPhysicalDeviceVulkan12Features device_features_1_2;
@@ -1212,7 +1215,7 @@ static int setup_queue_families(AVHWDevi
};
qf[i] = (VkQueueFamilyProperties2) {
.sType = VK_STRUCTURE_TYPE_QUEUE_FAMILY_PROPERTIES_2,
- .pNext = &qf_vid[i],
+ .pNext = p->vkctx.extensions & FF_VK_EXT_VIDEO_QUEUE ? &qf_vid[i] : NULL,
};
}
@@ -1642,6 +1645,7 @@ static int vulkan_device_init(AVHWDevice
FFVulkanFunctions *vk = &p->vkctx.vkfn;
VkQueueFamilyProperties2 *qf;
VkQueueFamilyVideoPropertiesKHR *qf_vid;
+ VkPhysicalDeviceExternalSemaphoreInfo ext_sem_props_info;
int graph_index, comp_index, tx_index, enc_index, dec_index;
/* Set device extension flags */
@@ -1687,6 +1691,24 @@ static int vulkan_device_init(AVHWDevice
return AVERROR_EXTERNAL;
}
+ ext_sem_props_info = (VkPhysicalDeviceExternalSemaphoreInfo) {
+ .sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_EXTERNAL_SEMAPHORE_INFO,
+ };
+
+ /* Opaque FD semaphore properties */
+ ext_sem_props_info.handleType =
+#ifdef _WIN32
+ IsWindows8OrGreater()
+ ? VK_EXTERNAL_SEMAPHORE_HANDLE_TYPE_OPAQUE_WIN32_BIT
+ : VK_EXTERNAL_SEMAPHORE_HANDLE_TYPE_OPAQUE_WIN32_KMT_BIT;
+#else
+ VK_EXTERNAL_SEMAPHORE_HANDLE_TYPE_OPAQUE_FD_BIT;
+#endif
+ p->ext_sem_props_opaque.sType = VK_STRUCTURE_TYPE_EXTERNAL_SEMAPHORE_PROPERTIES;
+ vk->GetPhysicalDeviceExternalSemaphoreProperties(hwctx->phys_dev,
+ &ext_sem_props_info,
+ &p->ext_sem_props_opaque);
+
qf = av_malloc_array(qf_num, sizeof(VkQueueFamilyProperties2));
if (!qf)
return AVERROR(ENOMEM);
@@ -1703,7 +1725,7 @@ static int vulkan_device_init(AVHWDevice
};
qf[i] = (VkQueueFamilyProperties2) {
.sType = VK_STRUCTURE_TYPE_QUEUE_FAMILY_PROPERTIES_2,
- .pNext = &qf_vid[i],
+ .pNext = p->vkctx.extensions & FF_VK_EXT_VIDEO_QUEUE ? &qf_vid[i] : NULL,
};
}
@@ -1980,7 +2002,7 @@ static int vulkan_frames_get_constraints
count += vkfmt_from_pixfmt2(ctx, vk_formats_list[i].pixfmt,
p->use_linear_images ? VK_IMAGE_TILING_LINEAR :
VK_IMAGE_TILING_OPTIMAL,
- NULL, NULL, NULL, NULL, 0, 0) >= 0;
+ NULL, NULL, NULL, NULL, p->disable_multiplane, 1) >= 0;
}
constraints->valid_sw_formats = av_malloc_array(count + 1,
@@ -1993,7 +2015,7 @@ static int vulkan_frames_get_constraints
if (vkfmt_from_pixfmt2(ctx, vk_formats_list[i].pixfmt,
p->use_linear_images ? VK_IMAGE_TILING_LINEAR :
VK_IMAGE_TILING_OPTIMAL,
- NULL, NULL, NULL, NULL, 0, 0) >= 0) {
+ NULL, NULL, NULL, NULL, p->disable_multiplane, 1) >= 0) {
constraints->valid_sw_formats[count++] = vk_formats_list[i].pixfmt;
}
}
@@ -2346,8 +2368,19 @@ static int create_frame(AVHWFramesContex
VulkanDevicePriv *p = ctx->hwctx;
AVVulkanDeviceContext *hwctx = &p->p;
FFVulkanFunctions *vk = &p->vkctx.vkfn;
+ AVVkFrame *f;
- VkExportSemaphoreCreateInfo ext_sem_info = {
+ VkSemaphoreTypeCreateInfo sem_type_info = {
+ .sType = VK_STRUCTURE_TYPE_SEMAPHORE_TYPE_CREATE_INFO,
+ .semaphoreType = VK_SEMAPHORE_TYPE_TIMELINE,
+ .initialValue = 0,
+ };
+ VkSemaphoreCreateInfo sem_spawn = {
+ .sType = VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO,
+ .pNext = &sem_type_info,
+ };
+
+ VkExportSemaphoreCreateInfo ext_sem_info_opaque = {
.sType = VK_STRUCTURE_TYPE_EXPORT_SEMAPHORE_CREATE_INFO,
#ifdef _WIN32
.handleTypes = IsWindows8OrGreater()
@@ -2358,23 +2391,13 @@ static int create_frame(AVHWFramesContex
#endif
};
- VkSemaphoreTypeCreateInfo sem_type_info = {
- .sType = VK_STRUCTURE_TYPE_SEMAPHORE_TYPE_CREATE_INFO,
-#ifdef _WIN32
- .pNext = p->vkctx.extensions & FF_VK_EXT_EXTERNAL_WIN32_SEM ? &ext_sem_info : NULL,
-#else
- .pNext = p->vkctx.extensions & FF_VK_EXT_EXTERNAL_FD_SEM ? &ext_sem_info : NULL,
-#endif
- .semaphoreType = VK_SEMAPHORE_TYPE_TIMELINE,
- .initialValue = 0,
- };
-
- VkSemaphoreCreateInfo sem_spawn = {
- .sType = VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO,
- .pNext = &sem_type_info,
- };
+ /* Check if exporting is supported before chaining any structs */
+ if (p->ext_sem_props_opaque.externalSemaphoreFeatures & VK_EXTERNAL_SEMAPHORE_FEATURE_EXPORTABLE_BIT) {
+ if (p->vkctx.extensions & (FF_VK_EXT_EXTERNAL_WIN32_SEM | FF_VK_EXT_EXTERNAL_FD_SEM))
+ ff_vk_link_struct(&sem_type_info, &ext_sem_info_opaque);
+ }
- AVVkFrame *f = av_vk_frame_alloc();
+ f = av_vk_frame_alloc();
if (!f) {
av_log(ctx, AV_LOG_ERROR, "Unable to allocate memory for AVVkFrame!\n");
return AVERROR(ENOMEM);
@@ -2689,11 +2712,11 @@ static int vulkan_frames_init(AVHWFrames
!(hwctx->usage & VK_IMAGE_USAGE_VIDEO_DECODE_DST_BIT_KHR)));
int sampleable = hwctx->usage & (VK_IMAGE_USAGE_SAMPLED_BIT |
VK_IMAGE_USAGE_STORAGE_BIT);
+ hwctx->img_flags = VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT;
if (sampleable && !is_lone_dpb) {
- hwctx->img_flags = VK_IMAGE_CREATE_ALIAS_BIT;
+ hwctx->img_flags |= VK_IMAGE_CREATE_ALIAS_BIT;
if ((fmt->vk_planes > 1) && (hwctx->format[0] == fmt->vkf))
- hwctx->img_flags |= VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT |
- VK_IMAGE_CREATE_EXTENDED_USAGE_BIT;
+ hwctx->img_flags |= VK_IMAGE_CREATE_EXTENDED_USAGE_BIT;
}
}
Index: FFmpeg/libavutil/vulkan.c
===================================================================
--- FFmpeg.orig/libavutil/vulkan.c
+++ FFmpeg/libavutil/vulkan.c
@@ -185,7 +185,7 @@ int ff_vk_load_props(FFVulkanContext *s)
};
s->qf_props[i] = (VkQueueFamilyProperties2) {
.sType = VK_STRUCTURE_TYPE_QUEUE_FAMILY_PROPERTIES_2,
- .pNext = &s->video_props[i],
+ .pNext = s->extensions & FF_VK_EXT_VIDEO_QUEUE ? &s->video_props[i] : NULL,
};
}
Index: FFmpeg/libavutil/vulkan_functions.h
===================================================================
--- FFmpeg.orig/libavutil/vulkan_functions.h
+++ FFmpeg/libavutil/vulkan_functions.h
@@ -33,10 +33,8 @@ typedef enum FFVulkanExtensions {
FF_VK_EXT_EXTERNAL_FD_SEM = 1ULL << 3, /* VK_KHR_external_semaphore_fd */
FF_VK_EXT_EXTERNAL_HOST_MEMORY = 1ULL << 4, /* VK_EXT_external_memory_host */
FF_VK_EXT_DEBUG_UTILS = 1ULL << 5, /* VK_EXT_debug_utils */
-#ifdef _WIN32
FF_VK_EXT_EXTERNAL_WIN32_MEMORY = 1ULL << 6, /* VK_KHR_external_memory_win32 */
FF_VK_EXT_EXTERNAL_WIN32_SEM = 1ULL << 7, /* VK_KHR_external_semaphore_win32 */
-#endif
FF_VK_EXT_DESCRIPTOR_BUFFER = 1ULL << 8, /* VK_EXT_descriptor_buffer */
FF_VK_EXT_DEVICE_DRM = 1ULL << 9, /* VK_EXT_physical_device_drm */
FF_VK_EXT_VIDEO_QUEUE = 1ULL << 10, /* VK_KHR_video_queue */
@@ -75,6 +73,7 @@ typedef enum FFVulkanExtensions {
MACRO(1, 0, FF_VK_EXT_NO_FLAG, CreateDevice) \
MACRO(1, 0, FF_VK_EXT_NO_FLAG, GetPhysicalDeviceFeatures2) \
MACRO(1, 0, FF_VK_EXT_NO_FLAG, GetPhysicalDeviceProperties) \
+ MACRO(1, 0, FF_VK_EXT_NO_FLAG, GetPhysicalDeviceExternalSemaphoreProperties) \
MACRO(1, 0, FF_VK_EXT_VIDEO_QUEUE, GetPhysicalDeviceVideoCapabilitiesKHR) \
MACRO(1, 0, FF_VK_EXT_VIDEO_QUEUE, GetPhysicalDeviceVideoFormatPropertiesKHR) \
MACRO(1, 0, FF_VK_EXT_NO_FLAG, DeviceWaitIdle) \
@@ -55,7 +55,7 @@ Index: FFmpeg/libavcodec/bsf/av1_metadata.c
+ } + }
+ } + }
+ +
+ if (ctx->remove_hdr10plus && provider_code == ITU_T_T35_PROVIDER_CODE_SMTPE) { + if (ctx->remove_hdr10plus && provider_code == ITU_T_T35_PROVIDER_CODE_SAMSUNG) {
+ provider_oriented_code = AV_RB16(t35->payload + 2); + provider_oriented_code = AV_RB16(t35->payload + 2);
+ application_identifier = AV_RB8(t35->payload + 4); + application_identifier = AV_RB8(t35->payload + 4);
+ // HDR10+ Metadata + // HDR10+ Metadata
@@ -131,7 +131,7 @@ Index: FFmpeg/libavcodec/bsf/h265_metadata.c
+ uint8_t *nal_sei = au->units[i].data; + uint8_t *nal_sei = au->units[i].data;
+ // This Matches ITU-T T.35 SMPTE ST 2094-40 + // This Matches ITU-T T.35 SMPTE ST 2094-40
+ if (nal_sei[0] == 0x4E && nal_sei[1] == 0x01 && nal_sei[2] == 0x04) { + if (nal_sei[0] == 0x4E && nal_sei[1] == 0x01 && nal_sei[2] == 0x04) {
+ if (nal_sei[4] == ITU_T_T35_COUNTRY_CODE_US && nal_sei[6] == ITU_T_T35_PROVIDER_CODE_SMTPE) { + if (nal_sei[4] == ITU_T_T35_COUNTRY_CODE_US && nal_sei[6] == ITU_T_T35_PROVIDER_CODE_SAMSUNG) {
+ // identifier for HDR10+ + // identifier for HDR10+
+ const uint8_t smpte2094_40_provider_oriented_code = 0x01; + const uint8_t smpte2094_40_provider_oriented_code = 0x01;
+ const uint8_t smpte2094_40_application_identifier = 0x04; + const uint8_t smpte2094_40_application_identifier = 0x04;
@@ -1,624 +0,0 @@
Index: FFmpeg/libavutil/hwcontext_vulkan.c
===================================================================
--- FFmpeg.orig/libavutil/hwcontext_vulkan.c
+++ FFmpeg/libavutil/hwcontext_vulkan.c
@@ -139,9 +139,6 @@ typedef struct VulkanFramesPriv {
FFVkExecPool upload_exec;
FFVkExecPool download_exec;
- /* Temporary buffer pools */
- AVBufferPool *tmp;
-
/* Modifier info list to free at uninit */
VkImageDrmFormatModifierListCreateInfoEXT *modifier_info;
} VulkanFramesPriv;
@@ -2622,8 +2619,6 @@ static void vulkan_frames_uninit(AVHWFra
ff_vk_exec_pool_free(&p->vkctx, &fp->compute_exec);
ff_vk_exec_pool_free(&p->vkctx, &fp->upload_exec);
ff_vk_exec_pool_free(&p->vkctx, &fp->download_exec);
-
- av_buffer_pool_uninit(&fp->tmp);
}
static int vulkan_frames_init(AVHWFramesContext *hwfc)
@@ -3809,290 +3804,128 @@ static int vulkan_map_from(AVHWFramesCon
return AVERROR(ENOSYS);
}
-static int copy_buffer_data(AVHWFramesContext *hwfc, AVBufferRef *buf,
- AVFrame *swf, VkBufferImageCopy *region,
- int planes, int upload)
+static size_t get_req_buffer_size(VulkanDevicePriv *p, int *stride, int height)
{
- VkResult ret;
- VulkanDevicePriv *p = hwfc->device_ctx->hwctx;
- FFVulkanFunctions *vk = &p->vkctx.vkfn;
- AVVulkanDeviceContext *hwctx = &p->p;
-
- FFVkBuffer *vkbuf = (FFVkBuffer *)buf->data;
-
- const VkMappedMemoryRange flush_info = {
- .sType = VK_STRUCTURE_TYPE_MAPPED_MEMORY_RANGE,
- .memory = vkbuf->mem,
- .size = VK_WHOLE_SIZE,
- };
-
- if (!(vkbuf->flags & VK_MEMORY_PROPERTY_HOST_COHERENT_BIT) && !upload) {
- ret = vk->InvalidateMappedMemoryRanges(hwctx->act_dev, 1,
- &flush_info);
- if (ret != VK_SUCCESS) {
- av_log(hwfc, AV_LOG_ERROR, "Failed to invalidate buffer data: %s\n",
- ff_vk_ret2str(ret));
- return AVERROR_EXTERNAL;
- }
- }
-
- for (int i = 0; i < planes; i++)
- av_image_copy_plane(vkbuf->mapped_mem + region[i].bufferOffset,
- region[i].bufferRowLength,
- swf->data[i],
- swf->linesize[i],
- swf->linesize[i],
- region[i].imageExtent.height);
-
- if (!(vkbuf->flags & VK_MEMORY_PROPERTY_HOST_COHERENT_BIT) && upload) {
- ret = vk->FlushMappedMemoryRanges(hwctx->act_dev, 1,
- &flush_info);
- if (ret != VK_SUCCESS) {
- av_log(hwfc, AV_LOG_ERROR, "Failed to flush buffer data: %s\n",
- ff_vk_ret2str(ret));
- return AVERROR_EXTERNAL;
- }
- }
-
- return 0;
+ size_t size;
+ *stride = FFALIGN(*stride, p->props.properties.limits.optimalBufferCopyRowPitchAlignment);
+ size = height*(*stride);
+ size = FFALIGN(size, p->props.properties.limits.minMemoryMapAlignment);
+ return size;
}
-static int get_plane_buf(AVHWFramesContext *hwfc, AVBufferRef **dst,
- AVFrame *swf, VkBufferImageCopy *region, int upload)
+static int transfer_image_buf(AVHWFramesContext *hwfc, AVFrame *f,
+ AVBufferRef **bufs, size_t *buf_offsets,
+ const int *buf_stride, int w,
+ int h, enum AVPixelFormat pix_fmt, int to_buf)
{
int err;
+ AVVkFrame *frame = (AVVkFrame *)f->data[0];
VulkanFramesPriv *fp = hwfc->hwctx;
VulkanDevicePriv *p = hwfc->device_ctx->hwctx;
- const int planes = av_pix_fmt_count_planes(swf->format);
-
- size_t buf_offset = 0;
- for (int i = 0; i < planes; i++) {
- size_t size;
- ptrdiff_t linesize = swf->linesize[i];
-
- uint32_t p_w, p_h;
- get_plane_wh(&p_w, &p_h, swf->format, swf->width, swf->height, i);
+ FFVulkanFunctions *vk = &p->vkctx.vkfn;
+ VkImageMemoryBarrier2 img_bar[AV_NUM_DATA_POINTERS];
+ int nb_img_bar = 0;
- linesize = FFALIGN(linesize,
- p->props.properties.limits.optimalBufferCopyRowPitchAlignment);
- size = p_h*linesize;
-
- region[i] = (VkBufferImageCopy) {
- .bufferOffset = buf_offset,
- .bufferRowLength = linesize,
- .bufferImageHeight = p_h,
- .imageSubresource.layerCount = 1,
- .imageExtent = (VkExtent3D){ p_w, p_h, 1 },
- /* Rest of the fields adjusted/filled in later */
- };
+ const int nb_images = ff_vk_count_images(frame);
+ int pixfmt_planes = av_pix_fmt_count_planes(pix_fmt);
+ const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(pix_fmt);
- buf_offset = FFALIGN(buf_offset + size,
- p->props.properties.limits.optimalBufferCopyOffsetAlignment);
- }
+ VkCommandBuffer cmd_buf;
+ FFVkExecContext *exec = ff_vk_exec_get(to_buf ? &fp->download_exec :
+ &fp->upload_exec);
+ cmd_buf = exec->buf;
+ ff_vk_exec_start(&p->vkctx, exec);
- err = ff_vk_get_pooled_buffer(&p->vkctx, &fp->tmp, dst,
- VK_BUFFER_USAGE_TRANSFER_SRC_BIT |
- VK_BUFFER_USAGE_TRANSFER_DST_BIT,
- NULL, buf_offset,
- VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT |
- VK_MEMORY_PROPERTY_HOST_CACHED_BIT);
+ err = ff_vk_exec_add_dep_buf(&p->vkctx, exec, bufs, pixfmt_planes, 1);
if (err < 0)
return err;
- return 0;
-}
-
-static int create_mapped_buffer(AVHWFramesContext *hwfc,
- FFVkBuffer *vkb, VkBufferUsageFlags usage,
- size_t size,
- VkExternalMemoryBufferCreateInfo *create_desc,
- VkImportMemoryHostPointerInfoEXT *import_desc,
- VkMemoryHostPointerPropertiesEXT props)
-{
- int err;
- VkResult ret;
- VulkanDevicePriv *p = hwfc->device_ctx->hwctx;
- FFVulkanFunctions *vk = &p->vkctx.vkfn;
- AVVulkanDeviceContext *hwctx = &p->p;
-
- VkBufferCreateInfo buf_spawn = {
- .sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO,
- .pNext = create_desc,
- .usage = usage,
- .sharingMode = VK_SHARING_MODE_EXCLUSIVE,
- .size = size,
- };
- VkMemoryRequirements req = {
- .size = size,
- .alignment = p->hprops.minImportedHostPointerAlignment,
- .memoryTypeBits = props.memoryTypeBits,
- };
-
- err = ff_vk_alloc_mem(&p->vkctx, &req,
- VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT,
- import_desc, &vkb->flags, &vkb->mem);
+ err = ff_vk_exec_add_dep_frame(&p->vkctx, exec, f,
+ VK_PIPELINE_STAGE_2_ALL_COMMANDS_BIT,
+ VK_PIPELINE_STAGE_2_TRANSFER_BIT);
if (err < 0)
return err;
- ret = vk->CreateBuffer(hwctx->act_dev, &buf_spawn, hwctx->alloc, &vkb->buf);
- if (ret != VK_SUCCESS) {
- vk->FreeMemory(hwctx->act_dev, vkb->mem, hwctx->alloc);
- return AVERROR_EXTERNAL;
- }
-
- ret = vk->BindBufferMemory(hwctx->act_dev, vkb->buf, vkb->mem, 0);
- if (ret != VK_SUCCESS) {
- vk->FreeMemory(hwctx->act_dev, vkb->mem, hwctx->alloc);
- vk->DestroyBuffer(hwctx->act_dev, vkb->buf, hwctx->alloc);
- return AVERROR_EXTERNAL;
- }
-
- return 0;
-}
-
-static void destroy_avvkbuf(void *opaque, uint8_t *data)
-{
- FFVulkanContext *s = opaque;
- FFVkBuffer *buf = (FFVkBuffer *)data;
- ff_vk_free_buf(s, buf);
- av_free(buf);
-}
-
-static int host_map_frame(AVHWFramesContext *hwfc, AVBufferRef **dst, int *nb_bufs,
- AVFrame *swf, VkBufferImageCopy *region, int upload)
-{
- int err;
- VkResult ret;
- VulkanDevicePriv *p = hwfc->device_ctx->hwctx;
- FFVulkanFunctions *vk = &p->vkctx.vkfn;
- AVVulkanDeviceContext *hwctx = &p->p;
-
- const int planes = av_pix_fmt_count_planes(swf->format);
-
- VkExternalMemoryBufferCreateInfo create_desc = {
- .sType = VK_STRUCTURE_TYPE_EXTERNAL_MEMORY_BUFFER_CREATE_INFO,
- .handleTypes = VK_EXTERNAL_MEMORY_HANDLE_TYPE_HOST_ALLOCATION_BIT_EXT,
- };
- VkImportMemoryHostPointerInfoEXT import_desc = {
- .sType = VK_STRUCTURE_TYPE_IMPORT_MEMORY_HOST_POINTER_INFO_EXT,
- .handleType = VK_EXTERNAL_MEMORY_HANDLE_TYPE_HOST_ALLOCATION_BIT_EXT,
- };
- VkMemoryHostPointerPropertiesEXT props;
-
- for (int i = 0; i < planes; i++) {
- FFVkBuffer *vkb;
- uint32_t p_w, p_h;
- size_t offs;
- size_t buffer_size;
-
- /* We can't host map images with negative strides */
- if (swf->linesize[i] < 0) {
- err = AVERROR(EINVAL);
- goto fail;
- }
-
- get_plane_wh(&p_w, &p_h, swf->format, swf->width, swf->height, i);
-
- /* Get the previous point at which mapping was possible and use it */
- offs = (uintptr_t)swf->data[i] % p->hprops.minImportedHostPointerAlignment;
- import_desc.pHostPointer = swf->data[i] - offs;
+ ff_vk_frame_barrier(&p->vkctx, exec, f, img_bar, &nb_img_bar,
+ VK_PIPELINE_STAGE_2_ALL_COMMANDS_BIT,
+ VK_PIPELINE_STAGE_2_TRANSFER_BIT_KHR,
+ to_buf ? VK_ACCESS_TRANSFER_READ_BIT :
+ VK_ACCESS_TRANSFER_WRITE_BIT,
+ to_buf ? VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL :
+ VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL,
+ VK_QUEUE_FAMILY_IGNORED);
- props = (VkMemoryHostPointerPropertiesEXT) {
- VK_STRUCTURE_TYPE_MEMORY_HOST_POINTER_PROPERTIES_EXT,
- };
- ret = vk->GetMemoryHostPointerPropertiesEXT(hwctx->act_dev,
- import_desc.handleType,
- import_desc.pHostPointer,
- &props);
- if (!(ret == VK_SUCCESS && props.memoryTypeBits)) {
- err = AVERROR(EINVAL);
- goto fail;
- }
+ vk->CmdPipelineBarrier2(cmd_buf, &(VkDependencyInfo) {
+ .sType = VK_STRUCTURE_TYPE_DEPENDENCY_INFO,
+ .pImageMemoryBarriers = img_bar,
+ .imageMemoryBarrierCount = nb_img_bar,
+ });
- /* Buffer region for this plane */
- region[i] = (VkBufferImageCopy) {
- .bufferOffset = offs,
- .bufferRowLength = swf->linesize[i],
- .bufferImageHeight = p_h,
+ /* Schedule a copy for each plane */
+ for (int i = 0; i < pixfmt_planes; i++) {
+ int idx = FFMIN(i, nb_images - 1);
+ VkImageAspectFlags plane_aspect[] = { VK_IMAGE_ASPECT_COLOR_BIT,
+ VK_IMAGE_ASPECT_PLANE_0_BIT,
+ VK_IMAGE_ASPECT_PLANE_1_BIT,
+ VK_IMAGE_ASPECT_PLANE_2_BIT, };
+
+ FFVkBuffer *vkbuf = (FFVkBuffer *)bufs[i]->data;
+ VkBufferImageCopy buf_reg = {
+ .bufferOffset = buf_offsets[i],
+ .bufferRowLength = buf_stride[i] / desc->comp[i].step,
.imageSubresource.layerCount = 1,
- .imageExtent = (VkExtent3D){ p_w, p_h, 1 },
- /* Rest of the fields adjusted/filled in later */
+ .imageSubresource.aspectMask = plane_aspect[(pixfmt_planes != nb_images) +
+ i*(pixfmt_planes != nb_images)],
+ .imageOffset = { 0, 0, 0, },
};
- /* Add the offset at the start, which gets ignored */
- buffer_size = offs + swf->linesize[i]*p_h;
- buffer_size = FFALIGN(buffer_size, p->props.properties.limits.minMemoryMapAlignment);
- buffer_size = FFALIGN(buffer_size, p->hprops.minImportedHostPointerAlignment);
-
- /* Create a buffer */
- vkb = av_mallocz(sizeof(*vkb));
- if (!vkb) {
- err = AVERROR(ENOMEM);
- goto fail;
- }
-
- err = create_mapped_buffer(hwfc, vkb,
- upload ? VK_BUFFER_USAGE_TRANSFER_SRC_BIT :
- VK_BUFFER_USAGE_TRANSFER_DST_BIT,
- buffer_size, &create_desc, &import_desc,
- props);
- if (err < 0) {
- av_free(vkb);
- goto fail;
- }
+ uint32_t p_w, p_h;
+ get_plane_wh(&p_w, &p_h, pix_fmt, w, h, i);
- /* Create a ref */
- dst[*nb_bufs] = av_buffer_create((uint8_t *)vkb, sizeof(*vkb),
- destroy_avvkbuf, &p->vkctx, 0);
- if (!dst[*nb_bufs]) {
- destroy_avvkbuf(&p->vkctx, (uint8_t *)vkb);
- err = AVERROR(ENOMEM);
- goto fail;
- }
+ buf_reg.bufferImageHeight = p_h;
+ buf_reg.imageExtent = (VkExtent3D){ p_w, p_h, 1, };
- (*nb_bufs)++;
+ if (to_buf)
+ vk->CmdCopyImageToBuffer(cmd_buf, frame->img[idx],
+ img_bar[0].newLayout,
+ vkbuf->buf,
+ 1, &buf_reg);
+ else
+ vk->CmdCopyBufferToImage(cmd_buf, vkbuf->buf, frame->img[idx],
+ img_bar[0].newLayout,
+ 1, &buf_reg);
}
- return 0;
+ err = ff_vk_exec_submit(&p->vkctx, exec);
+ if (err < 0)
+ return err;
-fail:
- for (int i = 0; i < (*nb_bufs); i++)
- av_buffer_unref(&dst[i]);
- return err;
+ ff_vk_exec_wait(&p->vkctx, exec);
+
+ return 0;
}
-static int vulkan_transfer_frame(AVHWFramesContext *hwfc,
- AVFrame *swf, AVFrame *hwf,
- int upload)
+static int vulkan_transfer_data(AVHWFramesContext *hwfc, const AVFrame *vkf,
+ const AVFrame *swf, int from)
{
- int err;
- VulkanFramesPriv *fp = hwfc->hwctx;
- VulkanDevicePriv *p = hwfc->device_ctx->hwctx;
+ int err = 0;
+ VkResult ret;
+ AVHWDeviceContext *dev_ctx = hwfc->device_ctx;
+ VulkanDevicePriv *p = dev_ctx->hwctx;
+ AVVulkanDeviceContext *hwctx = &p->p;
FFVulkanFunctions *vk = &p->vkctx.vkfn;
- int host_mapped = 0;
-
- AVVkFrame *hwf_vk = (AVVkFrame *)hwf->data[0];
- VkBufferImageCopy region[AV_NUM_DATA_POINTERS]; // always one per plane
+ AVFrame tmp;
+ FFVkBuffer *vkbufs[AV_NUM_DATA_POINTERS];
+ AVBufferRef *bufs[AV_NUM_DATA_POINTERS] = { 0 };
+ size_t buf_offsets[AV_NUM_DATA_POINTERS] = { 0 };
+ uint32_t p_w, p_h;
const int planes = av_pix_fmt_count_planes(swf->format);
- const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(swf->format);
- const int nb_images = ff_vk_count_images(hwf_vk);
- static const VkImageAspectFlags plane_aspect[] = { VK_IMAGE_ASPECT_COLOR_BIT,
- VK_IMAGE_ASPECT_PLANE_0_BIT,
- VK_IMAGE_ASPECT_PLANE_1_BIT,
- VK_IMAGE_ASPECT_PLANE_2_BIT, };
- VkImageMemoryBarrier2 img_bar[AV_NUM_DATA_POINTERS];
- int nb_img_bar = 0;
-
- AVBufferRef *bufs[AV_NUM_DATA_POINTERS];
- int nb_bufs = 0;
+ int host_mapped[AV_NUM_DATA_POINTERS] = { 0 };
+ const int map_host = !!(p->vkctx.extensions & FF_VK_EXT_EXTERNAL_HOST_MEMORY);
- VkCommandBuffer cmd_buf;
- FFVkExecContext *exec;
-
- /* Sanity checking */
if ((swf->format != AV_PIX_FMT_NONE && !av_vkfmt_from_pixfmt(swf->format))) {
av_log(hwfc, AV_LOG_ERROR, "Unsupported software frame pixel format!\n");
return AVERROR(EINVAL);
@@ -4101,97 +3934,115 @@ static int vulkan_transfer_frame(AVHWFra
if (swf->width > hwfc->width || swf->height > hwfc->height)
return AVERROR(EINVAL);
- /* Setup buffers first */
- if (p->vkctx.extensions & FF_VK_EXT_EXTERNAL_HOST_MEMORY) {
- err = host_map_frame(hwfc, bufs, &nb_bufs, swf, region, upload);
- if (err >= 0)
- host_mapped = 1;
- }
+ /* Create buffers */
+ for (int i = 0; i < planes; i++) {
+ size_t req_size;
+
+ VkExternalMemoryBufferCreateInfo create_desc = {
+ .sType = VK_STRUCTURE_TYPE_EXTERNAL_MEMORY_BUFFER_CREATE_INFO,
+ .handleTypes = VK_EXTERNAL_MEMORY_HANDLE_TYPE_HOST_ALLOCATION_BIT_EXT,
+ };
+
+ VkImportMemoryHostPointerInfoEXT import_desc = {
+ .sType = VK_STRUCTURE_TYPE_IMPORT_MEMORY_HOST_POINTER_INFO_EXT,
+ .handleType = VK_EXTERNAL_MEMORY_HANDLE_TYPE_HOST_ALLOCATION_BIT_EXT,
+ };
+
+ VkMemoryHostPointerPropertiesEXT p_props = {
+ .sType = VK_STRUCTURE_TYPE_MEMORY_HOST_POINTER_PROPERTIES_EXT,
+ };
+
+ get_plane_wh(&p_w, &p_h, swf->format, swf->width, swf->height, i);
- if (!host_mapped) {
- err = get_plane_buf(hwfc, &bufs[0], swf, region, upload);
+ tmp.linesize[i] = FFABS(swf->linesize[i]);
+
+ /* Do not map images with a negative stride */
+ if (map_host && swf->linesize[i] > 0) {
+ size_t offs;
+ offs = (uintptr_t)swf->data[i] % p->hprops.minImportedHostPointerAlignment;
+ import_desc.pHostPointer = swf->data[i] - offs;
+
+ /* We have to compensate for the few extra bytes of padding we
+ * completely ignore at the start */
+ req_size = FFALIGN(offs + tmp.linesize[i] * p_h,
+ p->hprops.minImportedHostPointerAlignment);
+
+ ret = vk->GetMemoryHostPointerPropertiesEXT(hwctx->act_dev,
+ import_desc.handleType,
+ import_desc.pHostPointer,
+ &p_props);
+ if (ret == VK_SUCCESS && p_props.memoryTypeBits) {
+ host_mapped[i] = 1;
+ buf_offsets[i] = offs;
+ }
+ }
+
+ if (!host_mapped[i])
+ req_size = get_req_buffer_size(p, &tmp.linesize[i], p_h);
+
+ err = ff_vk_create_avbuf(&p->vkctx, &bufs[i], req_size,
+ host_mapped[i] ? &create_desc : NULL,
+ host_mapped[i] ? &import_desc : NULL,
+ from ? VK_BUFFER_USAGE_TRANSFER_DST_BIT :
+ VK_BUFFER_USAGE_TRANSFER_SRC_BIT,
+ VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT |
+ (host_mapped[i] ?
+ VK_MEMORY_PROPERTY_HOST_COHERENT_BIT : 0x0));
if (err < 0)
goto end;
- nb_bufs = 1;
- if (upload) {
- err = copy_buffer_data(hwfc, bufs[0], swf, region, planes, 1);
- if (err < 0)
- goto end;
- }
+ vkbufs[i] = (FFVkBuffer *)bufs[i]->data;
}
- exec = ff_vk_exec_get(&fp->upload_exec);
- cmd_buf = exec->buf;
+ if (!from) {
+ /* Map, copy image TO buffer (which then goes to the VkImage), unmap */
+ if ((err = ff_vk_map_buffers(&p->vkctx, vkbufs, tmp.data, planes, 0)))
+ goto end;
- ff_vk_exec_start(&p->vkctx, exec);
+ for (int i = 0; i < planes; i++) {
+ if (host_mapped[i])
+ continue;
- /* Prep destination Vulkan frame */
- err = ff_vk_exec_add_dep_frame(&p->vkctx, exec, hwf,
- VK_PIPELINE_STAGE_2_ALL_COMMANDS_BIT,
- VK_PIPELINE_STAGE_2_TRANSFER_BIT);
- if (err < 0)
- goto end;
+ get_plane_wh(&p_w, &p_h, swf->format, swf->width, swf->height, i);
- /* No need to declare buf deps for synchronous transfers */
- if (upload) {
- err = ff_vk_exec_add_dep_buf(&p->vkctx, exec, bufs, nb_bufs, 1);
- if (err < 0) {
- ff_vk_exec_discard_deps(&p->vkctx, exec);
- goto end;
+ av_image_copy_plane(tmp.data[i], tmp.linesize[i],
+ (const uint8_t *)swf->data[i], swf->linesize[i],
+ FFMIN(tmp.linesize[i], FFABS(swf->linesize[i])),
+ p_h);
}
+
+ if ((err = ff_vk_unmap_buffers(&p->vkctx, vkbufs, planes, 1)))
+ goto end;
}
- ff_vk_frame_barrier(&p->vkctx, exec, hwf, img_bar, &nb_img_bar,
- VK_PIPELINE_STAGE_2_ALL_COMMANDS_BIT,
- VK_PIPELINE_STAGE_2_TRANSFER_BIT_KHR,
- upload ? VK_ACCESS_TRANSFER_WRITE_BIT :
- VK_ACCESS_TRANSFER_READ_BIT,
- upload ? VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL :
- VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL,
- VK_QUEUE_FAMILY_IGNORED);
+ /* Copy buffers into/from image */
+ err = transfer_image_buf(hwfc, (AVFrame *)vkf, bufs, buf_offsets,
+ tmp.linesize, swf->width, swf->height, swf->format,
+ from);
+
+ if (from) {
+ /* Map, copy buffer (which came FROM the VkImage) to the frame, unmap */
+ if ((err = ff_vk_map_buffers(&p->vkctx, vkbufs, tmp.data, planes, 0)))
+ goto end;
- vk->CmdPipelineBarrier2(cmd_buf, &(VkDependencyInfo) {
- .sType = VK_STRUCTURE_TYPE_DEPENDENCY_INFO,
- .pImageMemoryBarriers = img_bar,
- .imageMemoryBarrierCount = nb_img_bar,
- });
+ for (int i = 0; i < planes; i++) {
+ if (host_mapped[i])
+ continue;
- for (int i = 0; i < planes; i++) {
- int buf_idx = FFMIN(i, (nb_bufs - 1));
- int img_idx = FFMIN(i, (nb_images - 1));
- FFVkBuffer *vkbuf = (FFVkBuffer *)bufs[buf_idx]->data;
-
- uint32_t orig_stride = region[i].bufferRowLength;
- region[i].bufferRowLength /= desc->comp[i].step;
- region[i].imageSubresource.aspectMask = plane_aspect[(planes != nb_images) +
- i*(planes != nb_images)];
-
- if (upload)
- vk->CmdCopyBufferToImage(cmd_buf, vkbuf->buf,
- hwf_vk->img[img_idx],
- img_bar[img_idx].newLayout,
- 1, &region[i]);
- else
- vk->CmdCopyImageToBuffer(cmd_buf, hwf_vk->img[img_idx],
- img_bar[img_idx].newLayout,
- vkbuf->buf,
- 1, &region[i]);
+ get_plane_wh(&p_w, &p_h, swf->format, swf->width, swf->height, i);
- region[i].bufferRowLength = orig_stride;
- }
+ av_image_copy_plane_uc_from(swf->data[i], swf->linesize[i],
+ (const uint8_t *)tmp.data[i], tmp.linesize[i],
+ FFMIN(tmp.linesize[i], FFABS(swf->linesize[i])),
+ p_h);
+ }
- err = ff_vk_exec_submit(&p->vkctx, exec);
- if (err < 0) {
- ff_vk_exec_discard_deps(&p->vkctx, exec);
- } else if (!upload) {
- ff_vk_exec_wait(&p->vkctx, exec);
- if (!host_mapped)
- err = copy_buffer_data(hwfc, bufs[0], swf, region, planes, 0);
+ if ((err = ff_vk_unmap_buffers(&p->vkctx, vkbufs, planes, 1)))
+ goto end;
}
end:
- for (int i = 0; i < nb_bufs; i++)
+ for (int i = 0; i < planes; i++)
av_buffer_unref(&bufs[i]);
return err;
@@ -4218,7 +4069,7 @@ static int vulkan_transfer_data_to(AVHWF
if (src->hw_frames_ctx)
return AVERROR(ENOSYS);
else
- return vulkan_transfer_frame(hwfc, (AVFrame *)src, dst, 1);
+ return vulkan_transfer_data(hwfc, dst, src, 0);
}
}
@@ -4335,7 +4186,7 @@ static int vulkan_transfer_data_from(AVH
if (dst->hw_frames_ctx)
return AVERROR(ENOSYS);
else
- return vulkan_transfer_frame(hwfc, dst, (AVFrame *)src, 0);
+ return vulkan_transfer_data(hwfc, src, dst, 1);
}
}
Index: FFmpeg/libavutil/vulkan.c
===================================================================
--- FFmpeg.orig/libavutil/vulkan.c
+++ FFmpeg/libavutil/vulkan.c
@@ -887,8 +887,11 @@ int ff_vk_alloc_mem(FFVulkanContext *s,
ret = vk->AllocateMemory(s->hwctx->act_dev, &alloc_info,
s->hwctx->alloc, mem);
- if (ret != VK_SUCCESS)
+ if (ret != VK_SUCCESS) {
+ av_log(s, AV_LOG_ERROR, "Failed to allocate memory: %s\n",
+ ff_vk_ret2str(ret));
return AVERROR(ENOMEM);
+ }
if (mem_flags)
*mem_flags |= s->mprops.memoryTypes[index].propertyFlags;
@@ -22,19 +22,17 @@ Index: FFmpeg/libavcodec/audiotoolboxenc.c
FFAT_ENC_CLASS(NAME) \ FFAT_ENC_CLASS(NAME) \
const FFCodec ff_##NAME##_at_encoder = { \ const FFCodec ff_##NAME##_at_encoder = { \
.p.name = #NAME "_at", \ .p.name = #NAME "_at", \
@@ -652,9 +652,10 @@ static const AVOption options[] = { @@ -653,7 +653,8 @@ static const AVOption options[] = {
AV_CODEC_CAP_ENCODER_FLUSH CAPS, \
.p.ch_layouts = CH_LAYOUTS, \
.p.sample_fmts = (const enum AVSampleFormat[]) { \
- AV_SAMPLE_FMT_S16, \
+ AV_SAMPLE_FMT_S16, AV_SAMPLE_FMT_S32,\
AV_SAMPLE_FMT_U8, AV_SAMPLE_FMT_NONE \
}, \
+ .p.supported_samplerates = SAMPLE_RATES, \
.p.profiles = PROFILES, \ .p.profiles = PROFILES, \
.p.wrapper_name = "at", \ .p.wrapper_name = "at", \
CODEC_CH_LAYOUTS_ARRAY(CH_LAYOUTS), \
- CODEC_SAMPLEFMTS(AV_SAMPLE_FMT_S16, AV_SAMPLE_FMT_U8), \
+ CODEC_SAMPLERATES_ARRAY(SAMPLE_RATES), \
+ CODEC_SAMPLEFMTS(AV_SAMPLE_FMT_S16, AV_SAMPLE_FMT_S32, AV_SAMPLE_FMT_U8), \
}; };
@@ -669,15 +670,25 @@ static const AVChannelLayout aac_at_ch_l
static const AVChannelLayout aac_at_ch_layouts[] = {
@@ -666,15 +667,25 @@ static const AVChannelLayout aac_at_ch_l
AV_CHANNEL_LAYOUT_6POINT0, AV_CHANNEL_LAYOUT_6POINT0,
AV_CHANNEL_LAYOUT_6POINT1, AV_CHANNEL_LAYOUT_6POINT1,
AV_CHANNEL_LAYOUT_7POINT0, AV_CHANNEL_LAYOUT_7POINT0,
@@ -46,13 +44,13 @@ Index: FFmpeg/libavcodec/audiotoolboxenc.c
}; };
-FFAT_ENC(aac, AV_CODEC_ID_AAC, aac_profiles, , aac_at_channel_layouts, aac_at_ch_layouts) -FFAT_ENC(aac, AV_CODEC_ID_AAC, aac_profiles, , aac_at_channel_layouts, aac_at_ch_layouts)
+static const int aac_at_sample_rates[9] = { +static const int aac_at_sample_rates[] = {
+ 48000, 44100, 32000,24000, 22050, 16000, 12000, 11025, 8000 + 48000, 44100, 32000,24000, 22050, 16000, 12000, 11025, 8000, 0
+}; +};
+ +
+static const int alac_at_sample_rates[16] = { +static const int alac_at_sample_rates[] = {
+ 384000, 352800, 192000, 176400, 96000, 88200, 64000, 48000, + 384000, 352800, 192000, 176400, 96000, 88200, 64000, 48000,
+ 44100, 32000, 24000, 22050, 16000, 12000, 11025, 8000 + 44100, 32000, 24000, 22050, 16000, 12000, 11025, 8000, 0
+}; +};
+ +
+FFAT_ENC(aac, AV_CODEC_ID_AAC, aac_profiles, , aac_at_channel_layouts, aac_at_ch_layouts, aac_at_sample_rates) +FFAT_ENC(aac, AV_CODEC_ID_AAC, aac_profiles, , aac_at_channel_layouts, aac_at_ch_layouts, aac_at_sample_rates)
@@ -1,320 +0,0 @@
Index: FFmpeg/libavutil/hwcontext_vulkan.c
===================================================================
--- FFmpeg.orig/libavutil/hwcontext_vulkan.c
+++ FFmpeg/libavutil/hwcontext_vulkan.c
@@ -64,6 +64,13 @@
#if HAVE_LINUX_DMA_BUF_H
#include <sys/ioctl.h>
#include <linux/dma-buf.h>
+#if 0 // ndef DMA_BUF_IOCTL_EXPORT_SYNC_FILE
+#define DMA_BUF_IOCTL_EXPORT_SYNC_FILE _IOWR(DMA_BUF_BASE, 2, struct dma_buf_export_sync_file)
+struct dma_buf_export_sync_file {
+ uint32_t flags;
+ int32_t fd;
+};
+#endif
#endif
#if CONFIG_CUDA
@@ -1400,6 +1407,7 @@ static void vulkan_device_uninit(AVHWDev
static int vulkan_device_create_internal(AVHWDeviceContext *ctx,
VulkanDeviceSelection *dev_select,
+ int use_linear_images,
int disable_multiplane,
AVDictionary *opts, int flags)
{
@@ -1599,10 +1607,14 @@ static int vulkan_device_create_internal
goto end;
}
- /* Tiled images setting, use them by default */
- opt_d = av_dict_get(opts, "linear_images", NULL, 0);
- if (opt_d)
- p->use_linear_images = strtol(opt_d->value, NULL, 10);
+ /* Tiled images setting, use them by default.
+ * The use_linear_images argument takes precedent over the option */
+ p->use_linear_images = use_linear_images;
+ if (!p->use_linear_images) {
+ opt_d = av_dict_get(opts, "linear_images", NULL, 0);
+ if (opt_d)
+ p->use_linear_images = strtol(opt_d->value, NULL, 10);
+ }
/*
* The disable_multiplane argument takes precedent over the option.
@@ -1882,7 +1894,7 @@ static int vulkan_device_create(AVHWDevi
}
}
- return vulkan_device_create_internal(ctx, &dev_select, 0, opts, flags);
+ return vulkan_device_create_internal(ctx, &dev_select, 0, 0, opts, flags);
}
static int vulkan_device_derive(AVHWDeviceContext *ctx,
@@ -1906,6 +1918,7 @@ static int vulkan_device_derive(AVHWDevi
};
#endif
const char *vendor;
+ int use_linear_images = 0;
#if VA_CHECK_VERSION(1, 15, 0)
vas = vaGetDisplayAttributes(dpy, &attr, 1);
@@ -1920,11 +1933,14 @@ static int vulkan_device_derive(AVHWDevi
return AVERROR_EXTERNAL;
}
- if (strstr(vendor, "AMD"))
+ if (strstr(vendor, "AMD")) {
dev_select.vendor_id = 0x1002;
+ use_linear_images = 1;
+ }
}
- return vulkan_device_create_internal(ctx, &dev_select, 0, opts, flags);
+ /* AMD VA-API encoders require linear contiguous (multiplane) surface */
+ return vulkan_device_create_internal(ctx, &dev_select, use_linear_images, 0, opts, flags);
}
#endif
#if CONFIG_LIBDRM
@@ -1933,6 +1949,7 @@ static int vulkan_device_derive(AVHWDevi
struct stat drm_node_info;
drmDevice *drm_dev_info;
AVDRMDeviceContext *src_hwctx = src_ctx->hwctx;
+ int use_linear_images = 0;
err = fstat(src_hwctx->fd, &drm_node_info);
if (err) {
@@ -1952,12 +1969,15 @@ static int vulkan_device_derive(AVHWDevi
return AVERROR_EXTERNAL;
}
- if (drm_dev_info->bustype == DRM_BUS_PCI)
+ if (drm_dev_info->bustype == DRM_BUS_PCI) {
dev_select.pci_device = drm_dev_info->deviceinfo.pci->device_id;
+ use_linear_images = drm_dev_info->deviceinfo.pci->vendor_id == 0x1002;
+ }
drmFreeDevice(&drm_dev_info);
- return vulkan_device_create_internal(ctx, &dev_select, 0, opts, flags);
+ /* AMD VA-API encoders require linear contiguous (multiplane) surface */
+ return vulkan_device_create_internal(ctx, &dev_select, use_linear_images, 0, opts, flags);
}
#endif
#if CONFIG_CUDA
@@ -1980,7 +2000,7 @@ static int vulkan_device_derive(AVHWDevi
* CUDA is not able to import multiplane images, so always derive a
* Vulkan device with multiplane disabled.
*/
- return vulkan_device_create_internal(ctx, &dev_select, 1, opts, flags);
+ return vulkan_device_create_internal(ctx, &dev_select, 0, 1, opts, flags);
}
#endif
default:
@@ -2876,6 +2896,7 @@ static int vulkan_map_from_drm_frame_des
const AVDRMFrameDescriptor *desc = (AVDRMFrameDescriptor *)src->data[0];
VkBindImageMemoryInfo bind_info[AV_DRM_MAX_PLANES];
VkBindImagePlaneMemoryInfo plane_info[AV_DRM_MAX_PLANES];
+ const int has_modifiers = !!(p->vkctx.extensions & FF_VK_EXT_DRM_MODIFIER_FLAGS);
for (int i = 0; i < desc->nb_layers; i++) {
if (drm_to_vulkan_fmt(desc->layers[i].format) == VK_FORMAT_UNDEFINED) {
@@ -2885,13 +2906,21 @@ static int vulkan_map_from_drm_frame_des
}
}
+ if (!has_modifiers &&
+ desc->objects[0].format_modifier != DRM_FORMAT_MOD_INVALID &&
+ desc->objects[0].format_modifier != DRM_FORMAT_MOD_LINEAR) {
+ av_log(ctx, AV_LOG_ERROR, "The driver can only import DRM frame with invalid/linear modifier!\n");
+ err = AVERROR_EXTERNAL;
+ goto fail;
+ }
+
if (!(f = av_vk_frame_alloc())) {
av_log(ctx, AV_LOG_ERROR, "Unable to allocate memory for AVVkFrame!\n");
err = AVERROR(ENOMEM);
goto fail;
}
- f->tiling = VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT;
+ f->tiling = has_modifiers ? VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT : VK_IMAGE_TILING_LINEAR;
for (int i = 0; i < desc->nb_layers; i++) {
const int planes = desc->layers[i].nb_planes;
@@ -2929,7 +2958,7 @@ static int vulkan_map_from_drm_frame_des
.mipLevels = 1,
.arrayLayers = 1,
.flags = 0x0,
- .tiling = VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT,
+ .tiling = f->tiling,
.initialLayout = VK_IMAGE_LAYOUT_UNDEFINED, /* specs say so */
.usage = 0x0, /* filled in below */
.samples = VK_SAMPLE_COUNT_1_BIT,
@@ -2970,7 +2999,7 @@ static int vulkan_map_from_drm_frame_des
fmt_props = (VkPhysicalDeviceImageFormatInfo2) {
.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_IMAGE_FORMAT_INFO_2,
- .pNext = &props_ext,
+ .pNext = has_modifiers ? &props_ext : NULL,
.format = create_info.format,
.type = create_info.imageType,
.tiling = create_info.tiling,
@@ -3145,7 +3174,7 @@ static int vulkan_map_from_drm_frame_syn
const AVDRMFrameDescriptor *desc = (AVDRMFrameDescriptor *)src->data[0];
-#ifdef DMA_BUF_IOCTL_EXPORT_SYNC_FILE
+#if 0 // def DMA_BUF_IOCTL_EXPORT_SYNC_FILE
if (p->vkctx.extensions & FF_VK_EXT_EXTERNAL_FD_SEM) {
VkCommandBuffer cmd_buf;
FFVkExecContext *exec;
@@ -3171,8 +3200,10 @@ static int vulkan_map_from_drm_frame_syn
if (ioctl(desc->objects[i].fd, DMA_BUF_IOCTL_EXPORT_SYNC_FILE,
&implicit_fd_info)) {
err = AVERROR(errno);
- av_log(hwctx, AV_LOG_ERROR, "Failed to retrieve implicit DRM sync file: %s\n",
+ av_log(hwctx, i ? AV_LOG_ERROR : AV_LOG_DEBUG, "Failed to retrieve implicit DRM sync file: %s\n",
av_err2str(err));
+ if (i == 0) /* Bail out from the plane0 if ioctl failed */
+ goto fallback;
for (; i >= 0; i--)
vk->DestroySemaphore(hwctx->act_dev, drm_sync_sem[i], hwctx->alloc);
return err;
@@ -3248,10 +3279,12 @@ static int vulkan_map_from_drm_frame_syn
} else
#endif
{
- AVVkFrame *f = (AVVkFrame *)dst->data[0];
- av_log(hwctx, AV_LOG_WARNING, "No support for synchronization when importing DMA-BUFs, "
- "image may be corrupted.\n");
- err = prepare_frame(hwfc, &fp->compute_exec, f, PREP_MODE_EXTERNAL_IMPORT);
+#if 0 // def DMA_BUF_IOCTL_EXPORT_SYNC_FILE
+fallback:
+#endif
+ av_log(hwctx, AV_LOG_DEBUG, "No support for synchronization when importing DMA-BUFs, "
+ "image may be corrupted.\n");
+ err = prepare_frame(hwfc, &fp->compute_exec, (AVVkFrame *)dst->data[0], PREP_MODE_EXTERNAL_IMPORT);
if (err)
return err;
}
@@ -3594,22 +3627,14 @@ fail:
static int vulkan_map_to(AVHWFramesContext *hwfc, AVFrame *dst,
const AVFrame *src, int flags)
{
- av_unused VulkanDevicePriv *p = hwfc->device_ctx->hwctx;
-
switch (src->format) {
#if CONFIG_LIBDRM
#if CONFIG_VAAPI
case AV_PIX_FMT_VAAPI:
- if (p->vkctx.extensions & FF_VK_EXT_DRM_MODIFIER_FLAGS)
- return vulkan_map_from_vaapi(hwfc, dst, src, flags);
- else
- return AVERROR(ENOSYS);
+ return vulkan_map_from_vaapi(hwfc, dst, src, flags);
#endif
case AV_PIX_FMT_DRM_PRIME:
- if (p->vkctx.extensions & FF_VK_EXT_DRM_MODIFIER_FLAGS)
- return vulkan_map_from_drm(hwfc, dst, src, flags);
- else
- return AVERROR(ENOSYS);
+ return vulkan_map_from_drm(hwfc, dst, src, flags);
#endif
default:
return AVERROR(ENOSYS);
@@ -3652,13 +3677,14 @@ static int vulkan_map_to_drm(AVHWFramesC
VulkanFramesPriv *fp = hwfc->hwctx;
AVVulkanFramesContext *hwfctx = &fp->p;
const int planes = av_pix_fmt_count_planes(hwfc->sw_format);
+ const int has_modifiers = !!(p->vkctx.extensions & FF_VK_EXT_DRM_MODIFIER_FLAGS);
VkImageDrmFormatModifierPropertiesEXT drm_mod = {
.sType = VK_STRUCTURE_TYPE_IMAGE_DRM_FORMAT_MODIFIER_PROPERTIES_EXT,
};
VkSemaphoreWaitInfo wait_info = {
.sType = VK_STRUCTURE_TYPE_SEMAPHORE_WAIT_INFO,
.flags = 0x0,
- .semaphoreCount = planes,
+ .semaphoreCount = p->disable_multiplane ? planes : 1,
};
AVDRMFrameDescriptor *drm_desc = av_mallocz(sizeof(*drm_desc));
@@ -3679,10 +3705,16 @@ static int vulkan_map_to_drm(AVHWFramesC
if (err < 0)
goto end;
- ret = vk->GetImageDrmFormatModifierPropertiesEXT(hwctx->act_dev, f->img[0],
- &drm_mod);
- if (ret != VK_SUCCESS) {
- av_log(hwfc, AV_LOG_ERROR, "Failed to retrieve DRM format modifier!\n");
+ if (has_modifiers) {
+ ret = vk->GetImageDrmFormatModifierPropertiesEXT(hwctx->act_dev, f->img[0],
+ &drm_mod);
+ if (ret != VK_SUCCESS) {
+ av_log(hwfc, AV_LOG_ERROR, "Failed to retrieve DRM format modifier!\n");
+ err = AVERROR_EXTERNAL;
+ goto end;
+ }
+ } else if (f->tiling != VK_IMAGE_TILING_LINEAR) {
+ av_log(hwfc, AV_LOG_ERROR, "The driver can only export linear images to DRM frame!\n");
err = AVERROR_EXTERNAL;
goto end;
}
@@ -3704,7 +3736,7 @@ static int vulkan_map_to_drm(AVHWFramesC
drm_desc->nb_objects++;
drm_desc->objects[i].size = f->size[i];
- drm_desc->objects[i].format_modifier = drm_mod.drmFormatModifier;
+ drm_desc->objects[i].format_modifier = has_modifiers ? drm_mod.drmFormatModifier : 0x0;
}
drm_desc->nb_layers = planes;
@@ -3715,6 +3747,14 @@ static int vulkan_map_to_drm(AVHWFramesC
};
VkFormat plane_vkfmt = av_vkfmt_from_pixfmt(hwfc->sw_format)[i];
+ if (drm_desc->nb_layers > 1 && !p->disable_multiplane) {
+ switch (i) {
+ case 0: sub.aspectMask = VK_IMAGE_ASPECT_PLANE_0_BIT; break;
+ case 1: sub.aspectMask = VK_IMAGE_ASPECT_PLANE_1_BIT; break;
+ case 2: sub.aspectMask = VK_IMAGE_ASPECT_PLANE_2_BIT; break;
+ }
+ }
+
drm_desc->layers[i].format = vulkan_fmt_to_drm(plane_vkfmt);
drm_desc->layers[i].nb_planes = 1;
@@ -3729,7 +3769,7 @@ static int vulkan_map_to_drm(AVHWFramesC
if (f->tiling == VK_IMAGE_TILING_OPTIMAL)
continue;
- vk->GetImageSubresourceLayout(hwctx->act_dev, f->img[i], &sub, &layout);
+ vk->GetImageSubresourceLayout(hwctx->act_dev, f->img[p->disable_multiplane ? i : 0], &sub, &layout);
drm_desc->layers[i].planes[0].offset = layout.offset;
drm_desc->layers[i].planes[0].pitch = layout.rowPitch;
@@ -3781,21 +3821,13 @@ fail:
static int vulkan_map_from(AVHWFramesContext *hwfc, AVFrame *dst,
const AVFrame *src, int flags)
{
- av_unused VulkanDevicePriv *p = hwfc->device_ctx->hwctx;
-
switch (dst->format) {
#if CONFIG_LIBDRM
case AV_PIX_FMT_DRM_PRIME:
- if (p->vkctx.extensions & FF_VK_EXT_DRM_MODIFIER_FLAGS)
- return vulkan_map_to_drm(hwfc, dst, src, flags);
- else
- return AVERROR(ENOSYS);
+ return vulkan_map_to_drm(hwfc, dst, src, flags);
#if CONFIG_VAAPI
case AV_PIX_FMT_VAAPI:
- if (p->vkctx.extensions & FF_VK_EXT_DRM_MODIFIER_FLAGS)
- return vulkan_map_to_vaapi(hwfc, dst, src, flags);
- else
- return AVERROR(ENOSYS);
+ return vulkan_map_to_vaapi(hwfc, dst, src, flags);
#endif
#endif
default:
@@ -1,365 +0,0 @@
Index: FFmpeg/libavfilter/vf_overlay_vulkan.c
===================================================================
--- FFmpeg.orig/libavfilter/vf_overlay_vulkan.c
+++ FFmpeg/libavfilter/vf_overlay_vulkan.c
@@ -33,9 +33,11 @@ typedef struct OverlayVulkanContext {
int initialized;
FFVulkanPipeline pl;
+ FFVulkanPipeline pl_pass;
FFVkExecPool e;
FFVkQueueFamilyCtx qf;
FFVkSPIRVShader shd;
+ FFVkSPIRVShader shd_pass;
VkSampler sampler;
/* Push constants / options */
@@ -48,6 +50,10 @@ typedef struct OverlayVulkanContext {
int overlay_y;
int overlay_w;
int overlay_h;
+
+ int opt_repeatlast;
+ int opt_shortest;
+ int opt_eof_action;
} OverlayVulkanContext;
static const char overlay_noalpha[] = {
@@ -84,15 +90,16 @@ static const char overlay_alpha[] = {
static av_cold int init_filter(AVFilterContext *ctx)
{
int err;
- uint8_t *spv_data;
+ uint8_t *spv_data, *spv_data_pass;
size_t spv_len;
void *spv_opaque = NULL;
+ void *spv_opaque_pass = NULL;
OverlayVulkanContext *s = ctx->priv;
FFVulkanContext *vkctx = &s->vkctx;
const int planes = av_pix_fmt_count_planes(s->vkctx.output_format);
const int ialpha = av_pix_fmt_desc_get(s->vkctx.input_format)->flags & AV_PIX_FMT_FLAG_ALPHA;
const AVPixFmtDescriptor *pix_desc = av_pix_fmt_desc_get(s->vkctx.output_format);
- FFVkSPIRVShader *shd = &s->shd;
+ FFVkSPIRVShader *shd;
FFVkSPIRVCompiler *spv;
FFVulkanDescriptorSetBinding *desc;
@@ -105,90 +112,143 @@ static av_cold int init_filter(AVFilterC
ff_vk_qf_init(vkctx, &s->qf, VK_QUEUE_COMPUTE_BIT);
RET(ff_vk_exec_pool_init(vkctx, &s->qf, &s->e, s->qf.nb_queues*4, 0, 0, 0, NULL));
RET(ff_vk_init_sampler(vkctx, &s->sampler, 1, VK_FILTER_NEAREST));
+
+ /* overlay_compute */
RET(ff_vk_shader_init(&s->pl, &s->shd, "overlay_compute",
VK_SHADER_STAGE_COMPUTE_BIT, 0));
+ {
+ shd = &s->shd;
+ ff_vk_shader_set_compute_sizes(shd, 32, 32, 1);
+
+ GLSLC(0, layout(push_constant, std430) uniform pushConstants { );
+ GLSLC(1, ivec2 o_offset[3]; );
+ GLSLC(1, ivec2 o_size[3]; );
+ GLSLC(0, }; );
+ GLSLC(0, );
+
+ ff_vk_add_push_constant(&s->pl, 0, sizeof(s->opts),
+ VK_SHADER_STAGE_COMPUTE_BIT);
+
+ desc = (FFVulkanDescriptorSetBinding []) {
+ {
+ .name = "main_img",
+ .type = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER,
+ .dimensions = 2,
+ .elems = planes,
+ .stages = VK_SHADER_STAGE_COMPUTE_BIT,
+ .samplers = DUP_SAMPLER(s->sampler),
+ },
+ {
+ .name = "overlay_img",
+ .type = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER,
+ .dimensions = 2,
+ .elems = planes,
+ .stages = VK_SHADER_STAGE_COMPUTE_BIT,
+ .samplers = DUP_SAMPLER(s->sampler),
+ },
+ {
+ .name = "output_img",
+ .type = VK_DESCRIPTOR_TYPE_STORAGE_IMAGE,
+ .mem_layout = ff_vk_shader_rep_fmt(s->vkctx.output_format),
+ .mem_quali = "writeonly",
+ .dimensions = 2,
+ .elems = planes,
+ .stages = VK_SHADER_STAGE_COMPUTE_BIT,
+ },
+ };
+
+ RET(ff_vk_pipeline_descriptor_set_add(vkctx, &s->pl, shd, desc, 3, 0, 0));
+
+ GLSLD( overlay_noalpha );
+ GLSLD( overlay_alpha );
+ GLSLC(0, void main() );
+ GLSLC(0, { );
+ GLSLC(1, ivec2 pos = ivec2(gl_GlobalInvocationID.xy); );
+ GLSLF(1, int planes = %i; ,planes);
+ GLSLC(1, for (int i = 0; i < planes; i++) { );
+ if (ialpha)
+ GLSLC(2, overlay_alpha_opaque(i, pos); );
+ else
+ GLSLC(2, overlay_noalpha(i, pos); );
+ GLSLC(1, } );
+ GLSLC(0, } );
+
+ RET(spv->compile_shader(spv, ctx, shd, &spv_data, &spv_len, "main",
+ &spv_opaque));
+ RET(ff_vk_shader_create(vkctx, shd, spv_data, spv_len, "main"));
+
+ RET(ff_vk_init_compute_pipeline(vkctx, &s->pl, shd));
+ RET(ff_vk_exec_pipeline_register(vkctx, &s->e, &s->pl));
+
+ s->opts.o_offset[0] = s->overlay_x;
+ s->opts.o_offset[1] = s->overlay_y;
+ s->opts.o_offset[2] = s->opts.o_offset[0] >> pix_desc->log2_chroma_w;
+ s->opts.o_offset[3] = s->opts.o_offset[1] >> pix_desc->log2_chroma_h;
+ s->opts.o_offset[4] = s->opts.o_offset[0] >> pix_desc->log2_chroma_w;
+ s->opts.o_offset[5] = s->opts.o_offset[1] >> pix_desc->log2_chroma_h;
+
+ s->opts.o_size[0] = s->overlay_w;
+ s->opts.o_size[1] = s->overlay_h;
+ s->opts.o_size[2] = s->opts.o_size[0] >> pix_desc->log2_chroma_w;
+ s->opts.o_size[3] = s->opts.o_size[1] >> pix_desc->log2_chroma_h;
+ s->opts.o_size[4] = s->opts.o_size[0] >> pix_desc->log2_chroma_w;
+ s->opts.o_size[5] = s->opts.o_size[1] >> pix_desc->log2_chroma_h;
+ }
- ff_vk_shader_set_compute_sizes(&s->shd, 32, 32, 1);
+ /* overlay_compute_pass */
+ RET(ff_vk_shader_init(&s->pl_pass, &s->shd_pass, "overlay_compute_pass",
+ VK_SHADER_STAGE_COMPUTE_BIT, 0));
+ {
+ shd = &s->shd_pass;
+ ff_vk_shader_set_compute_sizes(shd, 32, 32, 1);
+
+ desc = (FFVulkanDescriptorSetBinding []) {
+ {
+ .name = "main_img",
+ .type = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER,
+ .dimensions = 2,
+ .elems = planes,
+ .stages = VK_SHADER_STAGE_COMPUTE_BIT,
+ .samplers = DUP_SAMPLER(s->sampler),
+ },
+ {
+ .name = "output_img",
+ .type = VK_DESCRIPTOR_TYPE_STORAGE_IMAGE,
+ .mem_layout = ff_vk_shader_rep_fmt(s->vkctx.output_format),
+ .mem_quali = "writeonly",
+ .dimensions = 2,
+ .elems = planes,
+ .stages = VK_SHADER_STAGE_COMPUTE_BIT,
+ },
+ };
+
+ RET(ff_vk_pipeline_descriptor_set_add(vkctx, &s->pl_pass, shd, desc, 2, 0, 0));
+
+ GLSLC(0, void main() );
+ GLSLC(0, { );
+ GLSLC(1, ivec2 pos = ivec2(gl_GlobalInvocationID.xy); );
+ GLSLF(1, int planes = %i; ,planes);
+ GLSLC(1, for (int i = 0; i < planes; i++) { );
+ GLSLC(2, vec4 res = texture(main_img[i], pos); );
+ GLSLC(2, imageStore(output_img[i], pos, res); );
+ GLSLC(1, } );
+ GLSLC(0, } );
+
+ RET(spv->compile_shader(spv, ctx, shd, &spv_data_pass, &spv_len, "main",
+ &spv_opaque));
+ RET(ff_vk_shader_create(vkctx, shd, spv_data_pass, spv_len, "main"));
- GLSLC(0, layout(push_constant, std430) uniform pushConstants { );
- GLSLC(1, ivec2 o_offset[3]; );
- GLSLC(1, ivec2 o_size[3]; );
- GLSLC(0, }; );
- GLSLC(0, );
-
- ff_vk_add_push_constant(&s->pl, 0, sizeof(s->opts),
- VK_SHADER_STAGE_COMPUTE_BIT);
-
- desc = (FFVulkanDescriptorSetBinding []) {
- {
- .name = "main_img",
- .type = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER,
- .dimensions = 2,
- .elems = planes,
- .stages = VK_SHADER_STAGE_COMPUTE_BIT,
- .samplers = DUP_SAMPLER(s->sampler),
- },
- {
- .name = "overlay_img",
- .type = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER,
- .dimensions = 2,
- .elems = planes,
- .stages = VK_SHADER_STAGE_COMPUTE_BIT,
- .samplers = DUP_SAMPLER(s->sampler),
- },
- {
- .name = "output_img",
- .type = VK_DESCRIPTOR_TYPE_STORAGE_IMAGE,
- .mem_layout = ff_vk_shader_rep_fmt(s->vkctx.output_format),
- .mem_quali = "writeonly",
- .dimensions = 2,
- .elems = planes,
- .stages = VK_SHADER_STAGE_COMPUTE_BIT,
- },
- };
-
- RET(ff_vk_pipeline_descriptor_set_add(vkctx, &s->pl, shd, desc, 3, 0, 0));
-
- GLSLD( overlay_noalpha );
- GLSLD( overlay_alpha );
- GLSLC(0, void main() );
- GLSLC(0, { );
- GLSLC(1, ivec2 pos = ivec2(gl_GlobalInvocationID.xy); );
- GLSLF(1, int planes = %i; ,planes);
- GLSLC(1, for (int i = 0; i < planes; i++) { );
- if (ialpha)
- GLSLC(2, overlay_alpha_opaque(i, pos); );
- else
- GLSLC(2, overlay_noalpha(i, pos); );
- GLSLC(1, } );
- GLSLC(0, } );
-
- RET(spv->compile_shader(spv, ctx, shd, &spv_data, &spv_len, "main",
- &spv_opaque));
- RET(ff_vk_shader_create(vkctx, shd, spv_data, spv_len, "main"));
-
- RET(ff_vk_init_compute_pipeline(vkctx, &s->pl, shd));
- RET(ff_vk_exec_pipeline_register(vkctx, &s->e, &s->pl));
-
- s->opts.o_offset[0] = s->overlay_x;
- s->opts.o_offset[1] = s->overlay_y;
- s->opts.o_offset[2] = s->opts.o_offset[0] >> pix_desc->log2_chroma_w;
- s->opts.o_offset[3] = s->opts.o_offset[1] >> pix_desc->log2_chroma_h;
- s->opts.o_offset[4] = s->opts.o_offset[0] >> pix_desc->log2_chroma_w;
- s->opts.o_offset[5] = s->opts.o_offset[1] >> pix_desc->log2_chroma_h;
-
- s->opts.o_size[0] = s->overlay_w;
- s->opts.o_size[1] = s->overlay_h;
- s->opts.o_size[2] = s->opts.o_size[0] >> pix_desc->log2_chroma_w;
- s->opts.o_size[3] = s->opts.o_size[1] >> pix_desc->log2_chroma_h;
- s->opts.o_size[4] = s->opts.o_size[0] >> pix_desc->log2_chroma_w;
- s->opts.o_size[5] = s->opts.o_size[1] >> pix_desc->log2_chroma_h;
+ RET(ff_vk_init_compute_pipeline(vkctx, &s->pl_pass, shd));
+ RET(ff_vk_exec_pipeline_register(vkctx, &s->e, &s->pl_pass));
+ }
s->initialized = 1;
fail:
if (spv_opaque)
spv->free_shader(spv, &spv_opaque);
+ if (spv_opaque_pass)
+ spv->free_shader(spv, &spv_opaque_pass);
if (spv)
spv->uninit(&spv);
@@ -210,22 +270,11 @@ static int overlay_vulkan_blend(FFFrameS
if (err < 0)
goto fail;
- if (!input_main || !input_overlay)
- return 0;
-
- if (!s->initialized) {
- AVHWFramesContext *main_fc = (AVHWFramesContext*)input_main->hw_frames_ctx->data;
- AVHWFramesContext *overlay_fc = (AVHWFramesContext*)input_overlay->hw_frames_ctx->data;
- if (main_fc->sw_format != overlay_fc->sw_format) {
- av_log(ctx, AV_LOG_ERROR, "Mismatching sw formats!\n");
- return AVERROR(EINVAL);
- }
-
- s->overlay_w = input_overlay->width;
- s->overlay_h = input_overlay->height;
+ if (!input_main)
+ return AVERROR_BUG;
+ if (!s->initialized)
RET(init_filter(ctx));
- }
out = ff_get_video_buffer(outlink, outlink->w, outlink->h);
if (!out) {
@@ -233,9 +282,13 @@ static int overlay_vulkan_blend(FFFrameS
goto fail;
}
- RET(ff_vk_filter_process_Nin(&s->vkctx, &s->e, &s->pl,
- out, (AVFrame *[]){ input_main, input_overlay }, 2,
- s->sampler, &s->opts, sizeof(s->opts)));
+ if (input_overlay)
+ RET(ff_vk_filter_process_Nin(&s->vkctx, &s->e, &s->pl,
+ out, (AVFrame *[]){ input_main, input_overlay }, 2,
+ s->sampler, &s->opts, sizeof(s->opts)));
+ else /* passthrough */
+ RET(ff_vk_filter_process_simple(&s->vkctx, &s->e, &s->pl_pass, out, input_main,
+ s->sampler, NULL, 0));
err = av_frame_copy_props(out, input_main);
if (err < 0)
@@ -253,6 +306,20 @@ static int overlay_vulkan_config_output(
int err;
AVFilterContext *avctx = outlink->src;
OverlayVulkanContext *s = avctx->priv;
+ AVFilterLink *inlink = avctx->inputs[0];
+ AVFilterLink *inlink_overlay = avctx->inputs[1];
+ FilterLink *inl = ff_filter_link(inlink);
+ FilterLink *inl_overlay = ff_filter_link(inlink_overlay);
+ AVHWFramesContext *main_fc = (AVHWFramesContext*)inl->hw_frames_ctx->data;
+ AVHWFramesContext *overlay_fc = (AVHWFramesContext*)inl_overlay->hw_frames_ctx->data;
+
+ if (main_fc->sw_format != overlay_fc->sw_format) {
+ av_log(avctx, AV_LOG_ERROR, "Mismatching sw formats!\n");
+ return AVERROR(EINVAL);
+ }
+
+ s->overlay_w = inlink_overlay->w;
+ s->overlay_h = inlink_overlay->h;
err = ff_vk_filter_config_output(outlink);
if (err < 0)
@@ -262,6 +329,11 @@ static int overlay_vulkan_config_output(
if (err < 0)
return err;
+ s->fs.opt_repeatlast = s->opt_repeatlast;
+ s->fs.opt_shortest = s->opt_shortest;
+ s->fs.opt_eof_action = s->opt_eof_action;
+ s->fs.time_base = outlink->time_base = inlink->time_base;
+
return ff_framesync_configure(&s->fs);
}
@@ -289,7 +361,9 @@ static void overlay_vulkan_uninit(AVFilt
ff_vk_exec_pool_free(vkctx, &s->e);
ff_vk_pipeline_free(vkctx, &s->pl);
+ ff_vk_pipeline_free(vkctx, &s->pl_pass);
ff_vk_shader_free(vkctx, &s->shd);
+ ff_vk_shader_free(vkctx, &s->shd_pass);
if (s->sampler)
vk->DestroySampler(vkctx->hwctx->act_dev, s->sampler,
@@ -306,6 +380,14 @@ static void overlay_vulkan_uninit(AVFilt
static const AVOption overlay_vulkan_options[] = {
{ "x", "Set horizontal offset", OFFSET(overlay_x), AV_OPT_TYPE_INT, {.i64 = 0}, 0, INT_MAX, .flags = FLAGS },
{ "y", "Set vertical offset", OFFSET(overlay_y), AV_OPT_TYPE_INT, {.i64 = 0}, 0, INT_MAX, .flags = FLAGS },
+ { "eof_action", "Action to take when encountering EOF from secondary input ",
+ OFFSET(opt_eof_action), AV_OPT_TYPE_INT, { .i64 = EOF_ACTION_REPEAT },
+ EOF_ACTION_REPEAT, EOF_ACTION_PASS, .flags = FLAGS, .unit = "eof_action" },
+ { "repeat", "Repeat the previous frame.", 0, AV_OPT_TYPE_CONST, { .i64 = EOF_ACTION_REPEAT }, .flags = FLAGS, .unit = "eof_action" },
+ { "endall", "End both streams.", 0, AV_OPT_TYPE_CONST, { .i64 = EOF_ACTION_ENDALL }, .flags = FLAGS, .unit = "eof_action" },
+ { "pass", "Pass through the main input.", 0, AV_OPT_TYPE_CONST, { .i64 = EOF_ACTION_PASS }, .flags = FLAGS, .unit = "eof_action" },
+ { "shortest", "force termination when the shortest input terminates", OFFSET(opt_shortest), AV_OPT_TYPE_BOOL, { .i64 = 0 }, 0, 1, FLAGS },
+ { "repeatlast", "repeat overlay of the last overlay frame", OFFSET(opt_repeatlast), AV_OPT_TYPE_BOOL, { .i64 = 1 }, 0, 1, FLAGS },
{ NULL },
};
@@ -1,217 +0,0 @@
Index: FFmpeg/libswscale/swscale_unscaled.c
===================================================================
--- FFmpeg.orig/libswscale/swscale_unscaled.c
+++ FFmpeg/libswscale/swscale_unscaled.c
@@ -396,7 +396,7 @@ static int planar8ToP01xleWrapper(SwsCon
const uint8_t *tsrc0 = src[0];
for (x = c->srcW; x > 0; x--) {
t = *tsrc0++;
- output_pixel(tdstY++, t | (t << 8));
+ output_pixel(tdstY++, (t << 8));
}
src[0] += srcStride[0];
dstY += dstStride[0] / 2;
@@ -407,9 +407,9 @@ static int planar8ToP01xleWrapper(SwsCon
const uint8_t *tsrc2 = src[2];
for (x = c->srcW / 2; x > 0; x--) {
t = *tsrc1++;
- output_pixel(tdstUV++, t | (t << 8));
+ output_pixel(tdstUV++, (t << 8));
t = *tsrc2++;
- output_pixel(tdstUV++, t | (t << 8));
+ output_pixel(tdstUV++, (t << 8));
}
src[1] += srcStride[1];
src[2] += srcStride[2];
Index: FFmpeg/tests/ref/fate/filter-pixdesc-p010le
===================================================================
--- FFmpeg.orig/tests/ref/fate/filter-pixdesc-p010le
+++ FFmpeg/tests/ref/fate/filter-pixdesc-p010le
@@ -1 +1 @@
-pixdesc-p010le 7b4a503997eb4e14cba80ee52db85e39
+pixdesc-p010le 0268fd44f63022e21ada69704534fc85
Index: FFmpeg/tests/ref/fate/filter-pixdesc-p016le
===================================================================
--- FFmpeg.orig/tests/ref/fate/filter-pixdesc-p016le
+++ FFmpeg/tests/ref/fate/filter-pixdesc-p016le
@@ -1 +1 @@
-pixdesc-p016le ed04897de0a6788bb3458e7365f10d36
+pixdesc-p016le 0268fd44f63022e21ada69704534fc85
Index: FFmpeg/tests/ref/fate/filter-pixfmts-copy
===================================================================
--- FFmpeg.orig/tests/ref/fate/filter-pixfmts-copy
+++ FFmpeg/tests/ref/fate/filter-pixfmts-copy
@@ -63,11 +63,11 @@ nv21 335d85c9af6110f26ae9
nv24 f30fc8d0ac40af69e119ea919a314572
nv42 29a212f70f8780fe0eb99abcae81894d
p010be 7f9842d6015026136bad60d03c035cc3
-p010le c453421b9f726bdaf2bacf59a492c43b
+p010le 1929db89609c4b8c6d9c9030a9e7843d
p012be 7f9842d6015026136bad60d03c035cc3
p012le 1929db89609c4b8c6d9c9030a9e7843d
p016be 7f9842d6015026136bad60d03c035cc3
-p016le c453421b9f726bdaf2bacf59a492c43b
+p016le 1929db89609c4b8c6d9c9030a9e7843d
p210be 847e9c6e292b17349e69570829252b3e
p210le c06e4b76cf504e908128081f92b60ce2
p212be 4df641ed058718ad27a01889f923b04f
Index: FFmpeg/tests/ref/fate/filter-pixfmts-crop
===================================================================
--- FFmpeg.orig/tests/ref/fate/filter-pixfmts-crop
+++ FFmpeg/tests/ref/fate/filter-pixfmts-crop
@@ -61,11 +61,11 @@ nv21 1bcfc197f4fb95de85ba
nv24 514c8f12082f0737e558778cbe7de258
nv42 ece9baae1c5de579dac2c66a89e08ef3
p010be 8b2de2eb6b099bbf355bfc55a0694ddc
-p010le 373b50c766dfd0a8e79c9a73246d803a
+p010le a1e4f713e145dfc465bfe0cc77096a03
p012be 8b2de2eb6b099bbf355bfc55a0694ddc
p012le a1e4f713e145dfc465bfe0cc77096a03
p016be 8b2de2eb6b099bbf355bfc55a0694ddc
-p016le 373b50c766dfd0a8e79c9a73246d803a
+p016le a1e4f713e145dfc465bfe0cc77096a03
p210be 2947f43774352ef61f9e83777548c7c5
p210le 74fcd5a32eee687eebe002c884103963
p212be c983aa869bae2c70e7b01810902ffc05
Index: FFmpeg/tests/ref/fate/filter-pixfmts-field
===================================================================
--- FFmpeg.orig/tests/ref/fate/filter-pixfmts-field
+++ FFmpeg/tests/ref/fate/filter-pixfmts-field
@@ -63,11 +63,11 @@ nv21 7294574037cc7f9373ef
nv24 3b100fb527b64ee2b2d7120da573faf5
nv42 1841ce853152d86b27c130f319ea0db2
p010be a0311a09bba7383553267d2b3b9c075e
-p010le ee09a18aefa3ebe97715b3a7312cb8ff
+p010le f1cc90d292046109a626db2da9f0f9b6
p012be a0311a09bba7383553267d2b3b9c075e
p012le f1cc90d292046109a626db2da9f0f9b6
p016be a0311a09bba7383553267d2b3b9c075e
-p016le ee09a18aefa3ebe97715b3a7312cb8ff
+p016le f1cc90d292046109a626db2da9f0f9b6
p210be 58d46f566ab28e3bcfb715c7aa53cf58
p210le 8d68f7655a3d76f2f8436bd25beb3973
p212be a8901966c5bc111e9e62d3989b0b666b
Index: FFmpeg/tests/ref/fate/filter-pixfmts-hflip
===================================================================
--- FFmpeg.orig/tests/ref/fate/filter-pixfmts-hflip
+++ FFmpeg/tests/ref/fate/filter-pixfmts-hflip
@@ -61,11 +61,11 @@ nv21 9f10dfff8963dc327d33
nv24 f0c5b2f42970f8d4003621d8857a872f
nv42 4dcf9aec82b110712b396a8b365dcb13
p010be 744b13e44d39e1ff7588983fa03e0101
-p010le a50b160346ab94f55a425065b57006f0
+p010le aeb31f50c66f376b0530c7bb6287212b
p012be 744b13e44d39e1ff7588983fa03e0101
p012le aeb31f50c66f376b0530c7bb6287212b
p016be 744b13e44d39e1ff7588983fa03e0101
-p016le a50b160346ab94f55a425065b57006f0
+p016le aeb31f50c66f376b0530c7bb6287212b
p210be 6f5a76d6467b86d55fe5589d3af8a7ea
p210le b6982912b2376371edea4fccf99fe40c
p212be 9ffa4664543233ec7c9b99a627cb7003
Index: FFmpeg/tests/ref/fate/filter-pixfmts-il
===================================================================
--- FFmpeg.orig/tests/ref/fate/filter-pixfmts-il
+++ FFmpeg/tests/ref/fate/filter-pixfmts-il
@@ -63,11 +63,11 @@ nv21 ab586d8781246b5a32d8
nv24 554153c71d142e3fd8e40b7dcaaec229
nv42 d699724c8deaeb4f87faf2766512eec3
p010be 3df51286ef66b53e3e283dbbab582263
-p010le eadcd8241e97e35b2b47d5eb2eaea6cd
+p010le 38945445b360fa737e9e37257393e823
p012be 3df51286ef66b53e3e283dbbab582263
p012le 38945445b360fa737e9e37257393e823
p016be 3df51286ef66b53e3e283dbbab582263
-p016le eadcd8241e97e35b2b47d5eb2eaea6cd
+p016le 38945445b360fa737e9e37257393e823
p210be 29ec4e8912d456cd15203a96487c42e8
p210le c695064fb9f2cc4e35957d4d649cc281
p212be ee6f88801823da3d617fb9e073e88068
Index: FFmpeg/tests/ref/fate/filter-pixfmts-null
===================================================================
--- FFmpeg.orig/tests/ref/fate/filter-pixfmts-null
+++ FFmpeg/tests/ref/fate/filter-pixfmts-null
@@ -63,11 +63,11 @@ nv21 335d85c9af6110f26ae9
nv24 f30fc8d0ac40af69e119ea919a314572
nv42 29a212f70f8780fe0eb99abcae81894d
p010be 7f9842d6015026136bad60d03c035cc3
-p010le c453421b9f726bdaf2bacf59a492c43b
+p010le 1929db89609c4b8c6d9c9030a9e7843d
p012be 7f9842d6015026136bad60d03c035cc3
p012le 1929db89609c4b8c6d9c9030a9e7843d
p016be 7f9842d6015026136bad60d03c035cc3
-p016le c453421b9f726bdaf2bacf59a492c43b
+p016le 1929db89609c4b8c6d9c9030a9e7843d
p210be 847e9c6e292b17349e69570829252b3e
p210le c06e4b76cf504e908128081f92b60ce2
p212be 4df641ed058718ad27a01889f923b04f
Index: FFmpeg/tests/ref/fate/filter-pixfmts-pad
===================================================================
--- FFmpeg.orig/tests/ref/fate/filter-pixfmts-pad
+++ FFmpeg/tests/ref/fate/filter-pixfmts-pad
@@ -28,9 +28,9 @@ nv16 d3a50501d2ea8535489f
nv21 0fdeb2cdd56cf5a7147dc273456fa217
nv24 193b9eadcc06ad5081609f76249b3e47
nv42 1738ad3c31c6c16e17679f5b09ce4677
-p010le fbbc23cc1d764a5e6fb71883d985f3ed
+p010le 3a92c1bd3e9de050bf6abcc3fd911ab7
p012le 3a92c1bd3e9de050bf6abcc3fd911ab7
-p016le fbbc23cc1d764a5e6fb71883d985f3ed
+p016le 3a92c1bd3e9de050bf6abcc3fd911ab7
p210le 680912c059de39c3401cac856bd1b0c1
p212le a2f88017bcce2383ba60bc4872e639ba
p216le 8718662e226a4581561e7bb532af2d83
Index: FFmpeg/tests/ref/fate/filter-pixfmts-scale
===================================================================
--- FFmpeg.orig/tests/ref/fate/filter-pixfmts-scale
+++ FFmpeg/tests/ref/fate/filter-pixfmts-scale
@@ -63,11 +63,11 @@ nv21 c74bb1c10dbbdee8a1f6
nv24 2aa6e805bf6d4179ed8d7dea37d75db3
nv42 80714d1eb2d8bcaeab3abc3124df1abd
p010be 1d6726d94bf1385996a9a9840dd0e878
-p010le 4b316f2b9e18972299beb73511278fa8
+p010le 5d436e6b35292a0e356d81f37f989b66
p012be e4dc7ccd654c2d74fde9c7b2711d960b
p012le cd4b6bdcd8967fc0e869ce3b8a014133
p016be 31e204018cbb53f8988c4e1174ea8ce9
-p016le d5afe557f492a09317e525d7cb782f5b
+p016le 6832661b5fe5f9a7a882f482a881b679
p210be 2cc6dfcf5e006c8ed5238988a06fd45e
p210le 04efb8f14a9d98417af40954a06aa187
p212be 611c6e267e7a694ce89467779e44060b
Index: FFmpeg/tests/ref/fate/filter-pixfmts-transpose
===================================================================
--- FFmpeg.orig/tests/ref/fate/filter-pixfmts-transpose
+++ FFmpeg/tests/ref/fate/filter-pixfmts-transpose
@@ -60,11 +60,11 @@ nv21 292adaf5271c5c8516b7
nv24 ea9de8b47faed722ee40182f89489beb
nv42 636af6cd6a4f3ac5edc0fc3ce3c56d63
p010be ad0de2cc9bff81688b182a870fcf7000
-p010le e7ff5143595021246733ce6bd0a769e8
+p010le 024ef1cf56a4872f202b96a6a4bbf10a
p012be ad0de2cc9bff81688b182a870fcf7000
p012le 024ef1cf56a4872f202b96a6a4bbf10a
p016be ad0de2cc9bff81688b182a870fcf7000
-p016le e7ff5143595021246733ce6bd0a769e8
+p016le 024ef1cf56a4872f202b96a6a4bbf10a
p410be 8b3e0ccb31b6a20ff00a29253fb2dec3
p410le 4e5f78dfccda9a6387e81354a56a033a
p412be 88e4578d2c6d99399a6cf1db9e4c0553
Index: FFmpeg/tests/ref/fate/filter-pixfmts-vflip
===================================================================
--- FFmpeg.orig/tests/ref/fate/filter-pixfmts-vflip
+++ FFmpeg/tests/ref/fate/filter-pixfmts-vflip
@@ -63,11 +63,11 @@ nv21 2909feacd27bebb080c8
nv24 334420b9d3df84499d2ca16bb66eed2b
nv42 ba4063e2795c17fea3c8a646b01fd1f5
p010be 06e9354b6e0e38ba41736352cedc0bd5
-p010le fd18d322bffbf5816902c13102872e22
+p010le cdf6a3c38d9d4e3f079fa369e1dda662
p012be 06e9354b6e0e38ba41736352cedc0bd5
p012le cdf6a3c38d9d4e3f079fa369e1dda662
p016be 06e9354b6e0e38ba41736352cedc0bd5
-p016le fd18d322bffbf5816902c13102872e22
+p016le cdf6a3c38d9d4e3f079fa369e1dda662
p210be ca886ab2b3ea5c153f1954b3709f7249
p210le d71c2d4e483030ffd87fa6a68c83fce0
p212be 1734e5840d4e75defe7a28683c3f8856
@@ -2,7 +2,7 @@ Index: FFmpeg/fftools/ffprobe.c
=================================================================== ===================================================================
--- FFmpeg.orig/fftools/ffprobe.c --- FFmpeg.orig/fftools/ffprobe.c
+++ FFmpeg/fftools/ffprobe.c +++ FFmpeg/fftools/ffprobe.c
@@ -147,6 +147,12 @@ static int show_private_data @@ -141,6 +141,12 @@ static const char *video_codec_name = NU
#define SHOW_OPTIONAL_FIELDS_ALWAYS 1 #define SHOW_OPTIONAL_FIELDS_ALWAYS 1
static int show_optional_fields = SHOW_OPTIONAL_FIELDS_AUTO; static int show_optional_fields = SHOW_OPTIONAL_FIELDS_AUTO;
@@ -15,7 +15,7 @@ Index: FFmpeg/fftools/ffprobe.c
static char *output_format; static char *output_format;
static char *stream_specifier; static char *stream_specifier;
static char *show_data_hash; static char *show_data_hash;
@@ -3108,9 +3114,13 @@ static int read_interval_packets(WriterC @@ -1596,9 +1602,13 @@ static int read_interval_packets(AVTextF
AVFormatContext *fmt_ctx = ifile->fmt_ctx; AVFormatContext *fmt_ctx = ifile->fmt_ctx;
AVPacket *pkt = NULL; AVPacket *pkt = NULL;
AVFrame *frame = NULL; AVFrame *frame = NULL;
@@ -30,7 +30,7 @@ Index: FFmpeg/fftools/ffprobe.c
av_log(NULL, AV_LOG_VERBOSE, "Processing read interval "); av_log(NULL, AV_LOG_VERBOSE, "Processing read interval ");
log_read_interval(interval, NULL, AV_LOG_VERBOSE); log_read_interval(interval, NULL, AV_LOG_VERBOSE);
@@ -3149,6 +3159,35 @@ static int read_interval_packets(WriterC @@ -1637,6 +1647,35 @@ static int read_interval_packets(AVTextF
ret = AVERROR(ENOMEM); ret = AVERROR(ENOMEM);
goto end; goto end;
} }
@@ -66,8 +66,8 @@ Index: FFmpeg/fftools/ffprobe.c
while (!av_read_frame(fmt_ctx, pkt)) { while (!av_read_frame(fmt_ctx, pkt)) {
if (fmt_ctx->nb_streams > nb_streams) { if (fmt_ctx->nb_streams > nb_streams) {
REALLOCZ_ARRAY_STREAM(nb_streams_frames, nb_streams, fmt_ctx->nb_streams); REALLOCZ_ARRAY_STREAM(nb_streams_frames, nb_streams, fmt_ctx->nb_streams);
@@ -3156,6 +3195,14 @@ static int read_interval_packets(WriterC @@ -1646,6 +1685,14 @@ static int read_interval_packets(AVTextF
REALLOCZ_ARRAY_STREAM(selected_streams, nb_streams, fmt_ctx->nb_streams); REALLOCZ_ARRAY_STREAM(streams_with_film_grain, nb_streams, fmt_ctx->nb_streams);
nb_streams = fmt_ctx->nb_streams; nb_streams = fmt_ctx->nb_streams;
} }
+ if (only_show_first_video_frame && nb_finished_video_streams < nb_video_streams && + if (only_show_first_video_frame && nb_finished_video_streams < nb_video_streams &&
@@ -81,7 +81,7 @@ Index: FFmpeg/fftools/ffprobe.c
if (selected_streams[pkt->stream_index]) { if (selected_streams[pkt->stream_index]) {
AVRational tb = ifile->streams[pkt->stream_index].st->time_base; AVRational tb = ifile->streams[pkt->stream_index].st->time_base;
int64_t pts = pkt->pts != AV_NOPTS_VALUE ? pkt->pts : pkt->dts; int64_t pts = pkt->pts != AV_NOPTS_VALUE ? pkt->pts : pkt->dts;
@@ -3181,6 +3228,28 @@ static int read_interval_packets(WriterC @@ -1671,6 +1718,28 @@ static int read_interval_packets(AVTextF
} }
frame_count++; frame_count++;
@@ -109,8 +109,8 @@ Index: FFmpeg/fftools/ffprobe.c
+ +
if (do_read_packets) { if (do_read_packets) {
if (do_show_packets) if (do_show_packets)
show_packet(w, ifile, pkt, i++); show_packet(tfc, ifile, pkt, i++);
@@ -3188,6 +3257,8 @@ static int read_interval_packets(WriterC @@ -1678,6 +1747,8 @@ static int read_interval_packets(AVTextF
} }
if (do_read_frames) { if (do_read_frames) {
int packet_new = 1; int packet_new = 1;
@@ -119,13 +119,13 @@ Index: FFmpeg/fftools/ffprobe.c
FrameData *fd; FrameData *fd;
pkt->opaque_ref = av_buffer_allocz(sizeof(*fd)); pkt->opaque_ref = av_buffer_allocz(sizeof(*fd));
@@ -3199,25 +3270,51 @@ static int read_interval_packets(WriterC @@ -1689,25 +1760,51 @@ static int read_interval_packets(AVTextF
fd->pkt_pos = pkt->pos; fd->pkt_pos = pkt->pos;
fd->pkt_size = pkt->size; fd->pkt_size = pkt->size;
- while (process_frame(w, ifile, frame, pkt, &packet_new) > 0); - while (process_frame(tfc, ifile, frame, pkt, &packet_new) > 0);
+ while (1) { + while (1) {
+ int frame_ret = process_frame(w, ifile, frame, pkt, &packet_new); + int frame_ret = process_frame(tfc, ifile, frame, pkt, &packet_new);
+ +
+ if (frame_ret <= 0) + if (frame_ret <= 0)
+ break; + break;
@@ -155,7 +155,7 @@ Index: FFmpeg/fftools/ffprobe.c
- for (i = 0; i < ifile->nb_streams; i++) { - for (i = 0; i < ifile->nb_streams; i++) {
- pkt->stream_index = i; - pkt->stream_index = i;
- if (do_read_frames) { - if (do_read_frames) {
- while (process_frame(w, ifile, frame, pkt, &(int){1}) > 0); - while (process_frame(tfc, ifile, frame, pkt, &(int){1}) > 0);
- if (ifile->streams[i].dec_ctx) - if (ifile->streams[i].dec_ctx)
- avcodec_flush_buffers(ifile->streams[i].dec_ctx); - avcodec_flush_buffers(ifile->streams[i].dec_ctx);
+ if (!only_show_first_video_frame) { + if (!only_show_first_video_frame) {
@@ -163,7 +163,7 @@ Index: FFmpeg/fftools/ffprobe.c
+ for (i = 0; i < ifile->nb_streams; i++) { + for (i = 0; i < ifile->nb_streams; i++) {
+ pkt->stream_index = i; + pkt->stream_index = i;
+ if (do_read_frames) { + if (do_read_frames) {
+ while (process_frame(w, ifile, frame, pkt, &(int){1}) > 0); + while (process_frame(tfc, ifile, frame, pkt, &(int){1}) > 0);
+ if (ifile->streams[i].dec_ctx) + if (ifile->streams[i].dec_ctx)
+ avcodec_flush_buffers(ifile->streams[i].dec_ctx); + avcodec_flush_buffers(ifile->streams[i].dec_ctx);
+ } + }
@@ -179,10 +179,10 @@ Index: FFmpeg/fftools/ffprobe.c
if (ret < 0) { if (ret < 0) {
av_log(NULL, AV_LOG_ERROR, "Could not read packets in interval "); av_log(NULL, AV_LOG_ERROR, "Could not read packets in interval ");
log_read_interval(interval, NULL, AV_LOG_ERROR); log_read_interval(interval, NULL, AV_LOG_ERROR);
@@ -4609,6 +4706,7 @@ static const OptionDef real_options[] = @@ -3208,6 +3305,7 @@ static const OptionDef real_options[] =
{ "print_filename", OPT_TYPE_FUNC, OPT_FUNC_ARG, {.func_arg = opt_print_filename}, "override the printed input filename", "print_file"},
{ "find_stream_info", OPT_TYPE_BOOL, OPT_INPUT | OPT_EXPERT, { &find_stream_info },
"read and decode the streams to fill missing information with heuristics" }, "read and decode the streams to fill missing information with heuristics" },
{ "c", OPT_TYPE_FUNC, OPT_FUNC_ARG, { .func_arg = opt_codec}, "force decoder", "decoder_name" },
{ "codec", OPT_TYPE_FUNC, OPT_FUNC_ARG, { .func_arg = opt_codec}, "alias for -c (force decoder)", "decoder_name" },
+ { "only_first_vframe", OPT_TYPE_BOOL, 0, { &only_show_first_video_frame }, "only show first video frame when show_frames is used" }, + { "only_first_vframe", OPT_TYPE_BOOL, 0, { &only_show_first_video_frame }, "only show first video frame when show_frames is used" },
{ NULL, }, { NULL, },
}; };
@@ -2,12 +2,13 @@ Index: FFmpeg/fftools/ffmpeg_filter.c
=================================================================== ===================================================================
--- FFmpeg.orig/fftools/ffmpeg_filter.c --- FFmpeg.orig/fftools/ffmpeg_filter.c
+++ FFmpeg/fftools/ffmpeg_filter.c +++ FFmpeg/fftools/ffmpeg_filter.c
@@ -1543,8 +1543,21 @@ static int configure_output_video_filter @@ -1708,9 +1708,22 @@ static int configure_output_video_filter
ofp->format != AV_PIX_FMT_NONE || !ofp->formats); ofp->format != AV_PIX_FMT_NONE || !ofp->pix_fmts);
av_bprint_init(&bprint, 0, AV_BPRINT_SIZE_UNLIMITED); av_bprint_init(&bprint, 0, AV_BPRINT_SIZE_UNLIMITED);
choose_pix_fmts(ofp, &bprint); choose_pix_fmts(ofp, &bprint);
- choose_color_spaces(ofp, &bprint); - choose_color_spaces(ofp, &bprint);
- choose_color_ranges(ofp, &bprint); - choose_color_ranges(ofp, &bprint);
- choose_alpha_modes(ofp, &bprint);
+ /* SW filter cannot handle color conversions between HW pixel formats. */ + /* SW filter cannot handle color conversions between HW pixel formats. */
+ { + {
+ int bprint_color_options = 1; + int bprint_color_options = 1;
@@ -21,6 +22,7 @@ Index: FFmpeg/fftools/ffmpeg_filter.c
+ if (bprint_color_options) { + if (bprint_color_options) {
+ choose_color_spaces(ofp, &bprint); + choose_color_spaces(ofp, &bprint);
+ choose_color_ranges(ofp, &bprint); + choose_color_ranges(ofp, &bprint);
+ choose_alpha_modes(ofp, &bprint);
+ } + }
+ } + }
if (!av_bprint_is_complete(&bprint)) if (!av_bprint_is_complete(&bprint))
@@ -14,7 +14,7 @@ Index: FFmpeg/fftools/ffmpeg_filter.c
+ ofp_fmt_desc = av_pix_fmt_desc_get(ofp->format); + ofp_fmt_desc = av_pix_fmt_desc_get(ofp->format);
+ // a dummy frame with HW pixel format requires a valid hw_frames_ctx + // a dummy frame with HW pixel format requires a valid hw_frames_ctx
+ if (ofp_fmt_desc->flags & AV_PIX_FMT_FLAG_HWACCEL) { + if (ofp_fmt_desc->flags & AV_PIX_FMT_FLAG_HWACCEL) {
+ AVFilterContext *sink = ofp->filter; + AVFilterContext *sink = ofp->ofilter.filter;
+ AVBufferRef *hwfc = av_buffersink_get_hw_frames_ctx(sink); + AVBufferRef *hwfc = av_buffersink_get_hw_frames_ctx(sink);
+ ret = av_buffer_replace(&frame->hw_frames_ctx, hwfc); + ret = av_buffer_replace(&frame->hw_frames_ctx, hwfc);
+ if (ret < 0) + if (ret < 0)
@@ -2,16 +2,16 @@ Index: FFmpeg/configure
=================================================================== ===================================================================
--- FFmpeg.orig/configure --- FFmpeg.orig/configure
+++ FFmpeg/configure +++ FFmpeg/configure
@@ -3867,6 +3867,7 @@ boxblur_opencl_filter_deps="opencl gpl" @@ -4121,6 +4121,7 @@ boxblur_opencl_filter_deps="opencl gpl"
bs2b_filter_deps="libbs2b" bs2b_filter_deps="libbs2b"
bwdif_cuda_filter_deps="ffnvcodec" bwdif_cuda_filter_deps="ffnvcodec"
bwdif_cuda_filter_deps_any="cuda_nvcc cuda_llvm" bwdif_cuda_filter_deps_any="cuda_nvcc cuda_llvm"
+bwdif_opencl_filter_deps="opencl" +bwdif_opencl_filter_deps="opencl"
bwdif_videotoolbox_filter_deps="metal corevideo videotoolbox" bwdif_videotoolbox_filter_deps="metal corevideo videotoolbox"
bwdif_vulkan_filter_deps="vulkan spirv_compiler" bwdif_vulkan_filter_deps="vulkan spirv_compiler"
chromaber_vulkan_filter_deps="vulkan spirv_compiler" chromaber_vulkan_filter_deps="vulkan spirv_library"
@@ -4019,6 +4020,7 @@ xfade_opencl_filter_deps="opencl" @@ -4280,6 +4281,7 @@ xfade_opencl_filter_deps="opencl"
xfade_vulkan_filter_deps="vulkan spirv_compiler" xfade_vulkan_filter_deps="vulkan spirv_library"
yadif_cuda_filter_deps="ffnvcodec" yadif_cuda_filter_deps="ffnvcodec"
yadif_cuda_filter_deps_any="cuda_nvcc cuda_llvm" yadif_cuda_filter_deps_any="cuda_nvcc cuda_llvm"
+yadif_opencl_filter_deps="opencl" +yadif_opencl_filter_deps="opencl"
@@ -22,7 +22,7 @@ Index: FFmpeg/libavfilter/Makefile
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/Makefile --- FFmpeg.orig/libavfilter/Makefile
+++ FFmpeg/libavfilter/Makefile +++ FFmpeg/libavfilter/Makefile
@@ -219,6 +219,8 @@ OBJS-$(CONFIG_BOXBLUR_OPENCL_FILTER) @@ -224,6 +224,8 @@ OBJS-$(CONFIG_BOXBLUR_OPENCL_FILTER)
OBJS-$(CONFIG_BWDIF_FILTER) += vf_bwdif.o bwdifdsp.o yadif_common.o OBJS-$(CONFIG_BWDIF_FILTER) += vf_bwdif.o bwdifdsp.o yadif_common.o
OBJS-$(CONFIG_BWDIF_CUDA_FILTER) += vf_bwdif_cuda.o vf_bwdif_cuda.ptx.o \ OBJS-$(CONFIG_BWDIF_CUDA_FILTER) += vf_bwdif_cuda.o vf_bwdif_cuda.ptx.o \
yadif_common.o yadif_common.o
@@ -31,7 +31,7 @@ Index: FFmpeg/libavfilter/Makefile
OBJS-$(CONFIG_BWDIF_VIDEOTOOLBOX_FILTER) += vf_bwdif_videotoolbox.o \ OBJS-$(CONFIG_BWDIF_VIDEOTOOLBOX_FILTER) += vf_bwdif_videotoolbox.o \
metal/vf_bwdif_videotoolbox.metallib.o \ metal/vf_bwdif_videotoolbox.metallib.o \
metal/utils.o \ metal/utils.o \
@@ -591,6 +593,8 @@ OBJS-$(CONFIG_XSTACK_FILTER) @@ -609,6 +611,8 @@ OBJS-$(CONFIG_XSTACK_FILTER)
OBJS-$(CONFIG_YADIF_FILTER) += vf_yadif.o yadif_common.o OBJS-$(CONFIG_YADIF_FILTER) += vf_yadif.o yadif_common.o
OBJS-$(CONFIG_YADIF_CUDA_FILTER) += vf_yadif_cuda.o vf_yadif_cuda.ptx.o \ OBJS-$(CONFIG_YADIF_CUDA_FILTER) += vf_yadif_cuda.o vf_yadif_cuda.ptx.o \
yadif_common.o cuda/load_helper.o yadif_common.o cuda/load_helper.o
@@ -44,22 +44,22 @@ Index: FFmpeg/libavfilter/allfilters.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/allfilters.c --- FFmpeg.orig/libavfilter/allfilters.c
+++ FFmpeg/libavfilter/allfilters.c +++ FFmpeg/libavfilter/allfilters.c
@@ -201,6 +201,7 @@ extern const AVFilter ff_vf_boxblur; @@ -204,6 +204,7 @@ extern const FFFilter ff_vf_boxblur;
extern const AVFilter ff_vf_boxblur_opencl; extern const FFFilter ff_vf_boxblur_opencl;
extern const AVFilter ff_vf_bwdif; extern const FFFilter ff_vf_bwdif;
extern const AVFilter ff_vf_bwdif_cuda; extern const FFFilter ff_vf_bwdif_cuda;
+extern const AVFilter ff_vf_bwdif_opencl; +extern const FFFilter ff_vf_bwdif_opencl;
extern const AVFilter ff_vf_bwdif_videotoolbox; extern const FFFilter ff_vf_bwdif_videotoolbox;
extern const AVFilter ff_vf_bwdif_vulkan; extern const FFFilter ff_vf_bwdif_vulkan;
extern const AVFilter ff_vf_cas; extern const FFFilter ff_vf_cas;
@@ -548,6 +549,7 @@ extern const AVFilter ff_vf_xpsnr; @@ -562,6 +563,7 @@ extern const FFFilter ff_vf_xpsnr;
extern const AVFilter ff_vf_xstack; extern const FFFilter ff_vf_xstack;
extern const AVFilter ff_vf_yadif; extern const FFFilter ff_vf_yadif;
extern const AVFilter ff_vf_yadif_cuda; extern const FFFilter ff_vf_yadif_cuda;
+extern const AVFilter ff_vf_yadif_opencl; +extern const FFFilter ff_vf_yadif_opencl;
extern const AVFilter ff_vf_yadif_videotoolbox; extern const FFFilter ff_vf_yadif_videotoolbox;
extern const AVFilter ff_vf_yaepblur; extern const FFFilter ff_vf_yaepblur;
extern const AVFilter ff_vf_zmq; extern const FFFilter ff_vf_zmq;
Index: FFmpeg/libavfilter/opencl.c Index: FFmpeg/libavfilter/opencl.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavfilter/opencl.c --- FFmpeg.orig/libavfilter/opencl.c
@@ -583,7 +583,7 @@ Index: FFmpeg/libavfilter/vf_bwdif_opencl.c
+ * License along with FFmpeg; if not, write to the Free Software + * License along with FFmpeg; if not, write to the Free Software
+ * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA + * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
+ */ + */
+ +
+#include "libavutil/avassert.h" +#include "libavutil/avassert.h"
+#include "libavutil/common.h" +#include "libavutil/common.h"
+#include "libavutil/imgutils.h" +#include "libavutil/imgutils.h"
@@ -878,17 +878,17 @@ Index: FFmpeg/libavfilter/vf_bwdif_opencl.c
+ }, + },
+}; +};
+ +
+const AVFilter ff_vf_bwdif_opencl = { +const FFFilter ff_vf_bwdif_opencl = {
+ .name = "bwdif_opencl", + .p.name = "bwdif_opencl",
+ .description = NULL_IF_CONFIG_SMALL("Deinterlace (BWDIF) the video through OpenCL."), + .p.description = NULL_IF_CONFIG_SMALL("Deinterlace (BWDIF) the video through OpenCL."),
+ .priv_size = sizeof(DeintOpenCLContext), + .priv_size = sizeof(DeintOpenCLContext),
+ .priv_class = &bwdif_opencl_class, + .p.priv_class = &bwdif_opencl_class,
+ .init = &deint_opencl_init, + .init = &deint_opencl_init,
+ .uninit = &deint_opencl_uninit, + .uninit = &deint_opencl_uninit,
+ FILTER_INPUTS(deint_opencl_inputs), + FILTER_INPUTS(deint_opencl_inputs),
+ FILTER_OUTPUTS(deint_opencl_outputs), + FILTER_OUTPUTS(deint_opencl_outputs),
+ FILTER_SINGLE_PIXFMT(AV_PIX_FMT_OPENCL), + FILTER_SINGLE_PIXFMT(AV_PIX_FMT_OPENCL),
+ .flags = AVFILTER_FLAG_HWDEVICE | + .p.flags = AVFILTER_FLAG_HWDEVICE |
+ AVFILTER_FLAG_SUPPORT_TIMELINE_INTERNAL, + AVFILTER_FLAG_SUPPORT_TIMELINE_INTERNAL,
+ .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE, + .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE,
+}; +};
@@ -916,7 +916,7 @@ Index: FFmpeg/libavfilter/vf_yadif_opencl.c
+ * License along with FFmpeg; if not, write to the Free Software + * License along with FFmpeg; if not, write to the Free Software
+ * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA + * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
+ */ + */
+ +
+#include "libavutil/avassert.h" +#include "libavutil/avassert.h"
+#include "libavutil/common.h" +#include "libavutil/common.h"
+#include "libavutil/imgutils.h" +#include "libavutil/imgutils.h"
@@ -1203,17 +1203,17 @@ Index: FFmpeg/libavfilter/vf_yadif_opencl.c
+ }, + },
+}; +};
+ +
+const AVFilter ff_vf_yadif_opencl = { +const FFFilter ff_vf_yadif_opencl = {
+ .name = "yadif_opencl", + .p.name = "yadif_opencl",
+ .description = NULL_IF_CONFIG_SMALL("Deinterlace (YADIF) the video through OpenCL."), + .p.description = NULL_IF_CONFIG_SMALL("Deinterlace (YADIF) the video through OpenCL."),
+ .priv_size = sizeof(DeintOpenCLContext), + .priv_size = sizeof(DeintOpenCLContext),
+ .priv_class = &yadif_opencl_class, + .p.priv_class = &yadif_opencl_class,
+ .init = &deint_opencl_init, + .init = &deint_opencl_init,
+ .uninit = &deint_opencl_uninit, + .uninit = &deint_opencl_uninit,
+ FILTER_INPUTS(deint_opencl_inputs), + FILTER_INPUTS(deint_opencl_inputs),
+ FILTER_OUTPUTS(deint_opencl_outputs), + FILTER_OUTPUTS(deint_opencl_outputs),
+ FILTER_SINGLE_PIXFMT(AV_PIX_FMT_OPENCL), + FILTER_SINGLE_PIXFMT(AV_PIX_FMT_OPENCL),
+ .flags = AVFILTER_FLAG_HWDEVICE | + .p.flags = AVFILTER_FLAG_HWDEVICE |
+ AVFILTER_FLAG_SUPPORT_TIMELINE_INTERNAL, + AVFILTER_FLAG_SUPPORT_TIMELINE_INTERNAL,
+ .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE, + .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE,
+}; +};
@@ -0,0 +1,159 @@
Index: FFmpeg/libavfilter/vf_tonemap_vaapi.c
===================================================================
--- FFmpeg.orig/libavfilter/vf_tonemap_vaapi.c
+++ FFmpeg/libavfilter/vf_tonemap_vaapi.c
@@ -56,7 +56,7 @@ static int tonemap_vaapi_save_metadata(A
AVContentLightMetadata *light_meta;
if (input_frame->color_trc != AVCOL_TRC_SMPTE2084) {
- av_log(avctx, AV_LOG_WARNING, "Only support HDR10 as input for vaapi tone-mapping\n");
+ av_log(avctx, AV_LOG_DEBUG, "Only support HDR10 as input for vaapi tone-mapping\n");
}
ctx->src_display = av_frame_get_side_data(input_frame,
@@ -122,8 +122,7 @@ static int tonemap_vaapi_save_metadata(A
ctx->in_metadata.white_point_y);
}
} else {
- av_log(avctx, AV_LOG_ERROR, "No mastering display data from input\n");
- return AVERROR(EINVAL);
+ av_log(avctx, AV_LOG_DEBUG, "No mastering display data from input\n");
}
ctx->src_light = av_frame_get_side_data(input_frame,
@@ -150,23 +149,20 @@ static int tonemap_vaapi_save_metadata(A
return 0;
}
-static int tonemap_vaapi_update_sidedata(AVFilterContext *avctx, AVFrame *output_frame)
+static int tonemap_vaapi_update_sidedata(AVFilterContext *avctx,
+ AVFrameSideData *metadata,
+ AVFrameSideData *metadata_lt)
{
HDRVAAPIContext *ctx = avctx->priv;
- AVFrameSideData *metadata;
AVMasteringDisplayMetadata *hdr_meta;
- AVFrameSideData *metadata_lt;
AVContentLightMetadata *hdr_meta_lt;
int i;
const int mapping[3] = {1, 2, 0}; //green, blue, red
const int chroma_den = 50000;
const int luma_den = 10000;
- metadata = av_frame_new_side_data(output_frame,
- AV_FRAME_DATA_MASTERING_DISPLAY_METADATA,
- sizeof(AVMasteringDisplayMetadata));
- if (!metadata)
- return AVERROR(ENOMEM);
+ if (!metadata || !metadata_lt)
+ return AVERROR(EINVAL);
hdr_meta = (AVMasteringDisplayMetadata *)metadata->data;
@@ -210,12 +206,6 @@ static int tonemap_vaapi_update_sidedata
ctx->out_metadata.max_display_mastering_luminance,
ctx->out_metadata.min_display_mastering_luminance);
- metadata_lt = av_frame_new_side_data(output_frame,
- AV_FRAME_DATA_CONTENT_LIGHT_LEVEL,
- sizeof(AVContentLightMetadata));
- if (!metadata_lt)
- return AVERROR(ENOMEM);
-
hdr_meta_lt = (AVContentLightMetadata *)metadata_lt->data;
hdr_meta_lt->MaxCLL = FFMIN(ctx->out_metadata.max_content_light_level, 65535);
@@ -396,7 +386,14 @@ static int tonemap_vaapi_filter_frame(AV
output_frame->colorspace = ctx->color_matrix;
if (ctx->mastering_display) {
- err = tonemap_vaapi_update_sidedata(avctx, output_frame);
+ AVFrameSideData *sd, *sd_lt;
+ sd = av_frame_new_side_data(output_frame,
+ AV_FRAME_DATA_MASTERING_DISPLAY_METADATA,
+ sizeof(AVMasteringDisplayMetadata));
+ sd_lt = av_frame_new_side_data(output_frame,
+ AV_FRAME_DATA_CONTENT_LIGHT_LEVEL,
+ sizeof(AVContentLightMetadata));
+ err = tonemap_vaapi_update_sidedata(avctx, sd, sd_lt);
if (err < 0)
goto fail;
}
@@ -457,6 +454,20 @@ static av_cold int tonemap_vaapi_init(AV
}
}
+ return 0;
+}
+
+static int tonemap_vaapi_config_output(AVFilterLink *outlink)
+{
+ AVFilterContext *avctx = outlink->src;
+ HDRVAAPIContext *ctx = avctx->priv;
+ int update_sd = 0, drop_sd = 0;
+ int err;
+
+ err = ff_vaapi_vpp_config_output(outlink);
+ if (err < 0)
+ return err;
+
#define STRING_OPTION(var_name, func_name, default_value) do { \
if (ctx->var_name ## _string) { \
int var = av_ ## func_name ## _from_name(ctx->var_name ## _string); \
@@ -505,6 +516,47 @@ static av_cold int tonemap_vaapi_init(AV
}
}
+ if (ctx->output_format_string) {
+ enum AVPixelFormat fmt = av_get_pix_fmt(ctx->output_format_string);
+ const AVPixFmtDescriptor *pix_desc = av_pix_fmt_desc_get(fmt);
+
+ if (!pix_desc || (pix_desc->flags & AV_PIX_FMT_FLAG_HWACCEL))
+ return AVERROR(EINVAL);
+
+ if (pix_desc->comp[0].depth < 10)
+ drop_sd = 1;
+ else if (ctx->mastering_display)
+ update_sd = 1;
+ else
+ drop_sd = 1;
+ } else {
+ if (ctx->mastering_display)
+ update_sd = 1;
+ else
+ drop_sd = 1;
+ }
+
+ if (drop_sd) {
+ av_frame_side_data_remove(&outlink->side_data, &outlink->nb_side_data,
+ AV_FRAME_DATA_MASTERING_DISPLAY_METADATA);
+ av_frame_side_data_remove(&outlink->side_data, &outlink->nb_side_data,
+ AV_FRAME_DATA_CONTENT_LIGHT_LEVEL);
+ }
+ if (update_sd) {
+ AVFrameSideData *sd, *sd_lt;
+ sd = av_frame_side_data_new(&outlink->side_data, &outlink->nb_side_data,
+ AV_FRAME_DATA_MASTERING_DISPLAY_METADATA,
+ sizeof(AVMasteringDisplayMetadata),
+ AV_FRAME_SIDE_DATA_FLAG_UNIQUE);
+ sd_lt = av_frame_side_data_new(&outlink->side_data, &outlink->nb_side_data,
+ AV_FRAME_DATA_CONTENT_LIGHT_LEVEL,
+ sizeof(AVContentLightMetadata),
+ AV_FRAME_SIDE_DATA_FLAG_UNIQUE);
+ err = tonemap_vaapi_update_sidedata(avctx, sd, sd_lt);
+ if (err < 0)
+ return err;
+ }
+
return 0;
}
@@ -555,7 +607,7 @@ static const AVFilterPad tonemap_vaapi_o
{
.name = "default",
.type = AVMEDIA_TYPE_VIDEO,
- .config_props = &ff_vaapi_vpp_config_output,
+ .config_props = &tonemap_vaapi_config_output,
},
};
@@ -232,7 +232,7 @@ Index: FFmpeg/libavcodec/dxva2.c
av_log((void *)avctx, AV_LOG_ERROR, "get_buffer frame is invalid!\n"); av_log((void *)avctx, AV_LOG_ERROR, "get_buffer frame is invalid!\n");
return NULL; return NULL;
} }
@@ -853,8 +932,10 @@ unsigned ff_dxva2_get_surface_index(cons @@ -852,8 +931,10 @@ unsigned ff_dxva2_get_surface_index(cons
} }
#endif #endif
#if CONFIG_D3D11VA #if CONFIG_D3D11VA
@@ -261,7 +261,7 @@ Index: FFmpeg/libavutil/hwcontext_d3d11va.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavutil/hwcontext_d3d11va.c --- FFmpeg.orig/libavutil/hwcontext_d3d11va.c
+++ FFmpeg/libavutil/hwcontext_d3d11va.c +++ FFmpeg/libavutil/hwcontext_d3d11va.c
@@ -318,7 +318,8 @@ static int d3d11va_frames_init(AVHWFrame @@ -329,7 +329,8 @@ static int d3d11va_frames_init(AVHWFrame
ctx->initial_pool_size = texDesc2.ArraySize; ctx->initial_pool_size = texDesc2.ArraySize;
hwctx->BindFlags = texDesc2.BindFlags; hwctx->BindFlags = texDesc2.BindFlags;
hwctx->MiscFlags = texDesc2.MiscFlags; hwctx->MiscFlags = texDesc2.MiscFlags;
@@ -275,7 +275,7 @@ Index: FFmpeg/libavutil/hwcontext_d3d11va.h
=================================================================== ===================================================================
--- FFmpeg.orig/libavutil/hwcontext_d3d11va.h --- FFmpeg.orig/libavutil/hwcontext_d3d11va.h
+++ FFmpeg/libavutil/hwcontext_d3d11va.h +++ FFmpeg/libavutil/hwcontext_d3d11va.h
@@ -188,6 +188,11 @@ typedef struct AVD3D11VAFramesContext { @@ -202,6 +202,11 @@ typedef struct AVD3D11VAFramesContext {
* Whether the frames require extra sync when exporting as external memory. * Whether the frames require extra sync when exporting as external memory.
*/ */
int require_sync; int require_sync;
@@ -0,0 +1,18 @@
Index: FFmpeg/libswscale/swscale.c
===================================================================
--- FFmpeg.orig/libswscale/swscale.c
+++ FFmpeg/libswscale/swscale.c
@@ -1430,7 +1430,12 @@ int sws_frame_setup(SwsContext *ctx, con
/* For now, if a single frame has a context, then both need a context */
if (!!src->hw_frames_ctx != !!dst->hw_frames_ctx) {
- return AVERROR(ENOTSUP);
+ const AVPixFmtDescriptor *src_desc = av_pix_fmt_desc_get(src->format);
+ const AVPixFmtDescriptor *dst_desc = av_pix_fmt_desc_get(dst->format);
+ const int is_mapped_hwframe_to_sw = !(src_desc->flags & AV_PIX_FMT_FLAG_HWACCEL) &&
+ !(dst_desc->flags & AV_PIX_FMT_FLAG_HWACCEL);
+ if (!is_mapped_hwframe_to_sw)
+ return AVERROR(ENOTSUP);
} else if (!!src->hw_frames_ctx) {
/* Both hardware frames must already be allocated */
if (!src->data[0] || !dst->data[0])
@@ -0,0 +1,101 @@
Index: FFmpeg/libavfilter/vf_hwupload.c
===================================================================
--- FFmpeg.orig/libavfilter/vf_hwupload.c
+++ FFmpeg/libavfilter/vf_hwupload.c
@@ -47,43 +47,35 @@ typedef struct HWUploadContext {
char *device_type;
} HWUploadContext;
-static int hwupload_init(AVFilterContext *avctx)
+static int hwupload_query_formats(AVFilterContext *avctx)
{
HWUploadContext *ctx = avctx->priv;
- int err;
+ AVHWFramesConstraints *constraints = NULL;
+ const enum AVPixelFormat *input_pix_fmts, *output_pix_fmts;
+ AVFilterFormats *input_formats = NULL;
+ int err, i;
- if (!avctx->hw_device_ctx) {
+ if (ctx->hwdevice_ref) {
+ /* We already have a specified device. */
+ } else if (avctx->hw_device_ctx) {
+ if (ctx->device_type) {
+ err = av_hwdevice_ctx_create_derived(
+ &ctx->hwdevice_ref,
+ av_hwdevice_find_type_by_name(ctx->device_type),
+ avctx->hw_device_ctx, 0);
+ if (err < 0)
+ return err;
+ } else {
+ ctx->hwdevice_ref = av_buffer_ref(avctx->hw_device_ctx);
+ if (!ctx->hwdevice_ref)
+ return AVERROR(ENOMEM);
+ }
+ } else {
av_log(ctx, AV_LOG_ERROR, "A hardware device reference is required "
"to upload frames to.\n");
return AVERROR(EINVAL);
}
- if (ctx->device_type) {
- err = av_hwdevice_ctx_create_derived(
- &ctx->hwdevice_ref,
- av_hwdevice_find_type_by_name(ctx->device_type),
- avctx->hw_device_ctx, 0);
- if (err < 0)
- return err;
- } else {
- ctx->hwdevice_ref = av_buffer_ref(avctx->hw_device_ctx);
- if (!ctx->hwdevice_ref)
- return AVERROR(ENOMEM);
- }
-
- return 0;
-}
-
-static int hwupload_query_formats(const AVFilterContext *avctx,
- AVFilterFormatsConfig **cfg_in,
- AVFilterFormatsConfig **cfg_out)
-{
- const HWUploadContext *ctx = avctx->priv;
- AVHWFramesConstraints *constraints = NULL;
- const enum AVPixelFormat *input_pix_fmts, *output_pix_fmts;
- AVFilterFormats *input_formats = NULL;
- int err, i;
-
constraints = av_hwdevice_get_hwframe_constraints(ctx->hwdevice_ref, NULL);
if (!constraints) {
err = AVERROR(EINVAL);
@@ -106,15 +98,16 @@ static int hwupload_query_formats(const
}
}
- if ((err = ff_formats_ref(input_formats, &cfg_in[0]->formats)) < 0 ||
+ if ((err = ff_formats_ref(input_formats, &avctx->inputs[0]->outcfg.formats)) < 0 ||
(err = ff_formats_ref(ff_make_pixel_format_list(output_pix_fmts),
- &cfg_out[0]->formats)) < 0)
+ &avctx->outputs[0]->incfg.formats)) < 0)
goto fail;
av_hwframe_constraints_free(&constraints);
return 0;
fail:
+ av_buffer_unref(&ctx->hwdevice_ref);
av_hwframe_constraints_free(&constraints);
return err;
}
@@ -287,11 +280,10 @@ const FFFilter ff_vf_hwupload = {
.p.description = NULL_IF_CONFIG_SMALL("Upload a normal frame to a hardware frame"),
.p.priv_class = &hwupload_class,
.p.flags = AVFILTER_FLAG_HWDEVICE,
- .init = hwupload_init,
.uninit = hwupload_uninit,
.priv_size = sizeof(HWUploadContext),
FILTER_INPUTS(hwupload_inputs),
FILTER_OUTPUTS(hwupload_outputs),
- FILTER_QUERY_FUNC2(hwupload_query_formats),
+ FILTER_QUERY_FUNC(hwupload_query_formats),
.flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE,
};
-354
View File
@@ -1,354 +0,0 @@
Index: FFmpeg/configure
===================================================================
--- FFmpeg.orig/configure
+++ FFmpeg/configure
@@ -2467,6 +2467,7 @@ TYPES_LIST="
kCMVideoCodecType_HEVC
kCMVideoCodecType_HEVCWithAlpha
kCMVideoCodecType_VP9
+ kCMVideoCodecType_AV1
kCVPixelFormatType_420YpCbCr10BiPlanarVideoRange
kCVPixelFormatType_422YpCbCr8BiPlanarVideoRange
kCVPixelFormatType_422YpCbCr10BiPlanarVideoRange
@@ -3176,6 +3177,8 @@ av1_vaapi_hwaccel_deps="vaapi VADecPictu
av1_vaapi_hwaccel_select="av1_decoder"
av1_vdpau_hwaccel_deps="vdpau VdpPictureInfoAV1"
av1_vdpau_hwaccel_select="av1_decoder"
+av1_videotoolbox_hwaccel_deps="videotoolbox"
+av1_videotoolbox_hwaccel_select="av1_decoder"
av1_vulkan_hwaccel_deps="vulkan"
av1_vulkan_hwaccel_select="av1_decoder"
h263_vaapi_hwaccel_deps="vaapi"
@@ -6744,6 +6747,7 @@ enabled videotoolbox && {
check_func_headers CoreMedia/CMFormatDescription.h kCMVideoCodecType_HEVC "-framework CoreMedia"
check_func_headers CoreMedia/CMFormatDescription.h kCMVideoCodecType_HEVCWithAlpha "-framework CoreMedia"
check_func_headers CoreMedia/CMFormatDescription.h kCMVideoCodecType_VP9 "-framework CoreMedia"
+ check_func_headers CoreMedia/CMFormatDescription.h kCMVideoCodecType_AV1 "-framework CoreMedia"
check_func_headers CoreVideo/CVPixelBuffer.h kCVPixelFormatType_420YpCbCr10BiPlanarVideoRange "-framework CoreVideo"
check_func_headers CoreVideo/CVPixelBuffer.h kCVPixelFormatType_422YpCbCr8BiPlanarVideoRange "-framework CoreVideo"
check_func_headers CoreVideo/CVPixelBuffer.h kCVPixelFormatType_422YpCbCr10BiPlanarVideoRange "-framework CoreVideo"
Index: FFmpeg/libavcodec/Makefile
===================================================================
--- FFmpeg.orig/libavcodec/Makefile
+++ FFmpeg/libavcodec/Makefile
@@ -1018,6 +1018,7 @@ OBJS-$(CONFIG_AV1_D3D12VA_HWACCEL)
OBJS-$(CONFIG_AV1_NVDEC_HWACCEL) += nvdec_av1.o
OBJS-$(CONFIG_AV1_VAAPI_HWACCEL) += vaapi_av1.o
OBJS-$(CONFIG_AV1_VDPAU_HWACCEL) += vdpau_av1.o
+OBJS-$(CONFIG_AV1_VIDEOTOOLBOX_HWACCEL) += videotoolbox_av1.o
OBJS-$(CONFIG_AV1_VULKAN_HWACCEL) += vulkan_decode.o vulkan_av1.o
OBJS-$(CONFIG_H263_VAAPI_HWACCEL) += vaapi_mpeg4.o
OBJS-$(CONFIG_H263_VIDEOTOOLBOX_HWACCEL) += videotoolbox.o
Index: FFmpeg/libavcodec/av1dec.c
===================================================================
--- FFmpeg.orig/libavcodec/av1dec.c
+++ FFmpeg/libavcodec/av1dec.c
@@ -541,6 +541,7 @@ static int get_pixel_format(AVCodecConte
CONFIG_AV1_NVDEC_HWACCEL + \
CONFIG_AV1_VAAPI_HWACCEL + \
CONFIG_AV1_VDPAU_HWACCEL + \
+ CONFIG_AV1_VIDEOTOOLBOX_HWACCEL + \
CONFIG_AV1_VULKAN_HWACCEL)
enum AVPixelFormat pix_fmts[HWACCEL_MAX + 2], *fmtp = pix_fmts;
@@ -568,6 +569,9 @@ static int get_pixel_format(AVCodecConte
#if CONFIG_AV1_VDPAU_HWACCEL
*fmtp++ = AV_PIX_FMT_VDPAU;
#endif
+#if CONFIG_AV1_VIDEOTOOLBOX_HWACCEL
+ *fmtp++ = AV_PIX_FMT_VIDEOTOOLBOX;
+#endif
#if CONFIG_AV1_VULKAN_HWACCEL
*fmtp++ = AV_PIX_FMT_VULKAN;
#endif
@@ -592,6 +596,9 @@ static int get_pixel_format(AVCodecConte
#if CONFIG_AV1_VDPAU_HWACCEL
*fmtp++ = AV_PIX_FMT_VDPAU;
#endif
+#if CONFIG_AV1_VIDEOTOOLBOX_HWACCEL
+ *fmtp++ = AV_PIX_FMT_VIDEOTOOLBOX;
+#endif
#if CONFIG_AV1_VULKAN_HWACCEL
*fmtp++ = AV_PIX_FMT_VULKAN;
#endif
@@ -1439,6 +1446,10 @@ static int av1_receive_frame_internal(AV
if (raw_tile_group && (s->tile_num == raw_tile_group->tg_end + 1)) {
int show_frame = s->raw_frame_header->show_frame;
+ // Set nb_unit to point at the next OBU, to indicate which
+ // OBUs have been processed for this current frame. (If this
+ // frame gets output, we set nb_unit to this value later too.)
+ s->nb_unit = i + 1;
if (avctx->hwaccel && s->cur_frame.f) {
ret = FF_HW_SIMPLE_CALL(avctx, end_frame);
if (ret < 0) {
@@ -1449,6 +1460,8 @@ static int av1_receive_frame_internal(AV
update_reference_list(avctx);
+ // Set start_unit to indicate the first OBU of the next frame.
+ s->start_unit = s->nb_unit;
raw_tile_group = NULL;
s->raw_frame_header = NULL;
@@ -1478,7 +1491,7 @@ end:
s->raw_frame_header = NULL;
av_packet_unref(s->pkt);
ff_cbs_fragment_reset(&s->current_obu);
- s->nb_unit = 0;
+ s->nb_unit = s->start_unit = 0;
}
if (!ret && !frame->buf[0])
ret = AVERROR(EAGAIN);
@@ -1505,7 +1518,7 @@ static int av1_receive_frame(AVCodecCont
return ret;
}
- s->nb_unit = 0;
+ s->nb_unit = s->start_unit = 0;
av_log(avctx, AV_LOG_DEBUG, "Total OBUs on this packet: %d.\n",
s->current_obu.nb_units);
}
@@ -1526,7 +1539,7 @@ static void av1_decode_flush(AVCodecCont
av1_frame_unref(&s->cur_frame);
s->operating_point_idc = 0;
- s->nb_unit = 0;
+ s->nb_unit = s->start_unit = 0;
s->raw_frame_header = NULL;
s->raw_seq = NULL;
s->cll = NULL;
@@ -1594,6 +1607,9 @@ const FFCodec ff_av1_decoder = {
#if CONFIG_AV1_VDPAU_HWACCEL
HWACCEL_VDPAU(av1),
#endif
+#if CONFIG_AV1_VIDEOTOOLBOX_HWACCEL
+ HWACCEL_VIDEOTOOLBOX(av1),
+#endif
#if CONFIG_AV1_VULKAN_HWACCEL
HWACCEL_VULKAN(av1),
#endif
Index: FFmpeg/libavcodec/av1dec.h
===================================================================
--- FFmpeg.orig/libavcodec/av1dec.h
+++ FFmpeg/libavcodec/av1dec.h
@@ -114,7 +114,8 @@ typedef struct AV1DecContext {
AV1Frame ref[AV1_NUM_REF_FRAMES];
AV1Frame cur_frame;
- int nb_unit;
+ int nb_unit; ///< The index of the next OBU to be processed.
+ int start_unit; ///< The index of the first OBU of the current frame.
// AVOptions
int operating_point;
Index: FFmpeg/libavcodec/hwaccels.h
===================================================================
--- FFmpeg.orig/libavcodec/hwaccels.h
+++ FFmpeg/libavcodec/hwaccels.h
@@ -26,6 +26,7 @@ extern const struct FFHWAccel ff_av1_dxv
extern const struct FFHWAccel ff_av1_nvdec_hwaccel;
extern const struct FFHWAccel ff_av1_vaapi_hwaccel;
extern const struct FFHWAccel ff_av1_vdpau_hwaccel;
+extern const struct FFHWAccel ff_av1_videotoolbox_hwaccel;
extern const struct FFHWAccel ff_av1_vulkan_hwaccel;
extern const struct FFHWAccel ff_h263_vaapi_hwaccel;
extern const struct FFHWAccel ff_h263_videotoolbox_hwaccel;
Index: FFmpeg/libavcodec/videotoolbox.c
===================================================================
--- FFmpeg.orig/libavcodec/videotoolbox.c
+++ FFmpeg/libavcodec/videotoolbox.c
@@ -56,6 +56,10 @@ enum { kCMVideoCodecType_HEVC = 'hvc1' }
enum { kCMVideoCodecType_VP9 = 'vp09' };
#endif
+#if !HAVE_KCMVIDEOCODECTYPE_AV1
+enum { kCMVideoCodecType_AV1 = 'av01' };
+#endif
+
#define VIDEOTOOLBOX_ESDS_EXTRADATA_PADDING 12
typedef struct VTHWFrame {
@@ -92,6 +96,26 @@ int ff_videotoolbox_buffer_copy(VTContex
return 0;
}
+int ff_videotoolbox_buffer_append(VTContext *vtctx,
+ const uint8_t *buffer,
+ uint32_t size)
+{
+ void *tmp;
+
+ tmp = av_fast_realloc(vtctx->bitstream,
+ &vtctx->allocated_size,
+ vtctx->bitstream_size + size);
+
+ if (!tmp)
+ return AVERROR(ENOMEM);
+
+ vtctx->bitstream = tmp;
+ memcpy(vtctx->bitstream + vtctx->bitstream_size, buffer, size);
+ vtctx->bitstream_size += size;
+
+ return 0;
+}
+
static int videotoolbox_postproc_frame(void *avctx, AVFrame *frame)
{
int ret;
@@ -840,6 +864,13 @@ static CFDictionaryRef videotoolbox_deco
CFDictionarySetValue(avc_info, CFSTR("vpcC"), data);
break;
#endif
+#if CONFIG_AV1_VIDEOTOOLBOX_HWACCEL
+ case kCMVideoCodecType_AV1 :
+ data = ff_videotoolbox_av1c_extradata_create(avctx);
+ if (data)
+ CFDictionarySetValue(avc_info, CFSTR("av1C"), data);
+ break;
+#endif
default:
break;
}
@@ -905,6 +936,9 @@ static int videotoolbox_start(AVCodecCon
case AV_CODEC_ID_VP9 :
videotoolbox->cm_codec_type = kCMVideoCodecType_VP9;
break;
+ case AV_CODEC_ID_AV1 :
+ videotoolbox->cm_codec_type = kCMVideoCodecType_AV1;
+ break;
default :
break;
}
Index: FFmpeg/libavcodec/videotoolbox_av1.c
===================================================================
--- /dev/null
+++ FFmpeg/libavcodec/videotoolbox_av1.c
@@ -0,0 +1,105 @@
+/*
+ * Videotoolbox hardware acceleration for AV1
+ * Copyright (c) 2023 Jan Ekström
+ * Copyright (c) 2024 Ruslan Chernenko
+ * Copyright (c) 2024 Martin Storsjö
+ *
+ * This file is part of FFmpeg.
+ *
+ * FFmpeg is free software; you can redistribute it and/or
+ * modify it under the terms of the GNU Lesser General Public
+ * License as published by the Free Software Foundation; either
+ * version 2.1 of the License, or (at your option) any later version.
+ *
+ * FFmpeg is distributed in the hope that it will be useful,
+ * but WITHOUT ANY WARRANTY; without even the implied warranty of
+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
+ * Lesser General Public License for more details.
+ *
+ * You should have received a copy of the GNU Lesser General Public
+ * License along with FFmpeg; if not, write to the Free Software
+ * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
+ */
+
+#include "libavutil/mem.h"
+
+#include "av1dec.h"
+#include "hwaccel_internal.h"
+#include "internal.h"
+#include "vt_internal.h"
+
+CFDataRef ff_videotoolbox_av1c_extradata_create(AVCodecContext *avctx)
+{
+ AV1DecContext *s = avctx->priv_data;
+ uint8_t *buf;
+ CFDataRef data;
+ if (!s->raw_seq)
+ return NULL;
+
+ buf = av_malloc(s->seq_data_ref->size + 4);
+ if (!buf)
+ return NULL;
+ buf[0] = 0x81; // version and marker (constant)
+ buf[1] = s->raw_seq->seq_profile << 5 | s->raw_seq->seq_level_idx[0];
+ buf[2] = s->raw_seq->seq_tier[0] << 7 |
+ s->raw_seq->color_config.high_bitdepth << 6 |
+ s->raw_seq->color_config.twelve_bit << 5 |
+ s->raw_seq->color_config.mono_chrome << 4 |
+ s->raw_seq->color_config.subsampling_x << 3 |
+ s->raw_seq->color_config.subsampling_y << 2 |
+ s->raw_seq->color_config.chroma_sample_position;
+
+ if (s->raw_seq->initial_display_delay_present_flag)
+ buf[3] = 0 << 5 |
+ s->raw_seq->initial_display_delay_present_flag << 4 |
+ s->raw_seq->initial_display_delay_minus_1[0];
+ else
+ buf[3] = 0x00;
+ memcpy(buf + 4, s->seq_data_ref->data, s->seq_data_ref->size);
+ data = CFDataCreate(kCFAllocatorDefault, buf, s->seq_data_ref->size + 4);
+ av_free(buf);
+ return data;
+};
+
+
+static int videotoolbox_av1_start_frame(AVCodecContext *avctx,
+ const uint8_t *buffer,
+ uint32_t size)
+{
+ return 0;
+}
+
+static int videotoolbox_av1_decode_slice(AVCodecContext *avctx,
+ const uint8_t *buffer,
+ uint32_t size)
+{
+ return 0;
+}
+
+static int videotoolbox_av1_end_frame(AVCodecContext *avctx)
+{
+ const AV1DecContext *s = avctx->priv_data;
+ VTContext *vtctx = avctx->internal->hwaccel_priv_data;
+ AVFrame *frame = s->cur_frame.f;
+
+ vtctx->bitstream_size = 0;
+ for (int i = s->start_unit; i < s->nb_unit; i++)
+ ff_videotoolbox_buffer_append(vtctx, s->current_obu.units[i].data,
+ s->current_obu.units[i].data_size);
+ return ff_videotoolbox_common_end_frame(avctx, frame);
+}
+
+const FFHWAccel ff_av1_videotoolbox_hwaccel = {
+ .p.name = "av1_videotoolbox",
+ .p.type = AVMEDIA_TYPE_VIDEO,
+ .p.id = AV_CODEC_ID_AV1,
+ .p.pix_fmt = AV_PIX_FMT_VIDEOTOOLBOX,
+ .alloc_frame = ff_videotoolbox_alloc_frame,
+ .start_frame = videotoolbox_av1_start_frame,
+ .decode_slice = videotoolbox_av1_decode_slice,
+ .end_frame = videotoolbox_av1_end_frame,
+ .frame_params = ff_videotoolbox_frame_params,
+ .init = ff_videotoolbox_common_init,
+ .uninit = ff_videotoolbox_uninit,
+ .priv_data_size = sizeof(VTContext),
+};
Index: FFmpeg/libavcodec/vt_internal.h
===================================================================
--- FFmpeg.orig/libavcodec/vt_internal.h
+++ FFmpeg/libavcodec/vt_internal.h
@@ -56,6 +56,9 @@ int ff_videotoolbox_frame_params(AVCodec
int ff_videotoolbox_buffer_copy(VTContext *vtctx,
const uint8_t *buffer,
uint32_t size);
+int ff_videotoolbox_buffer_append(VTContext *vtctx,
+ const uint8_t *buffer,
+ uint32_t size);
int ff_videotoolbox_uninit(AVCodecContext *avctx);
int ff_videotoolbox_h264_start_frame(AVCodecContext *avctx,
const uint8_t *buffer,
@@ -64,6 +67,7 @@ int ff_videotoolbox_h264_decode_slice(AV
const uint8_t *buffer,
uint32_t size);
int ff_videotoolbox_common_end_frame(AVCodecContext *avctx, AVFrame *frame);
+CFDataRef ff_videotoolbox_av1c_extradata_create(AVCodecContext *avctx);
CFDataRef ff_videotoolbox_avcc_extradata_create(AVCodecContext *avctx);
CFDataRef ff_videotoolbox_hvcc_extradata_create(AVCodecContext *avctx);
CFDataRef ff_videotoolbox_vpcc_extradata_create(AVCodecContext *avctx);
@@ -0,0 +1,13 @@
Index: FFmpeg/libavcodec/h264_slice.c
===================================================================
--- FFmpeg.orig/libavcodec/h264_slice.c
+++ FFmpeg/libavcodec/h264_slice.c
@@ -1150,7 +1150,7 @@ static int h264_init_ps(H264Context *h,
if (flush_changes)
ff_h264_flush_change(h);
- if ((ret = get_pixel_format(h, must_reinit || needs_reinit)) < 0)
+ if ((ret = get_pixel_format(h, 1)) < 0)
return ret;
h->avctx->pix_fmt = ret;
@@ -0,0 +1,22 @@
Index: FFmpeg/fftools/ffmpeg_demux.c
===================================================================
--- FFmpeg.orig/fftools/ffmpeg_demux.c
+++ FFmpeg/fftools/ffmpeg_demux.c
@@ -534,7 +534,7 @@ static void readrate_sleep(Demuxer *d)
ds->lag = lag;
ds->resume_wc = now;
ds->resume_pts = pts;
- av_log_once(ds, AV_LOG_WARNING, AV_LOG_DEBUG, &resume_warn,
+ av_log_once(ds, AV_LOG_VERBOSE, AV_LOG_DEBUG, &resume_warn,
"Resumed reading at pts %0.3f with rate %0.3f after a lag of %0.3fs\n",
(float)pts/AV_TIME_BASE, d->readrate_catchup, (float)lag/AV_TIME_BASE);
}
@@ -2152,7 +2152,7 @@ int ifile_open(const OptionsContext *o,
d->readrate_initial_burst);
return AVERROR(EINVAL);
}
- d->readrate_catchup = o->readrate_catchup ? o->readrate_catchup : d->readrate * 1.05;
+ d->readrate_catchup = o->readrate_catchup ? o->readrate_catchup : d->readrate * 100;
if (d->readrate_catchup < d->readrate) {
av_log(d, AV_LOG_ERROR,
"Option -readrate_catchup is %0.3f; it must be at least equal to %0.3f.\n",
@@ -0,0 +1,509 @@
Index: FFmpeg/libavcodec/vulkan_encode.c
===================================================================
--- FFmpeg.orig/libavcodec/vulkan_encode.c
+++ FFmpeg/libavcodec/vulkan_encode.c
@@ -772,14 +772,6 @@ av_cold int ff_vulkan_encode_init(AVCode
return AVERROR(EINVAL);
}
- if ((ctx->enc_caps.supportedEncodeFeedbackFlags & feedback_flags) !=
- feedback_flags) {
- av_log (avctx, AV_LOG_ERROR,
- "Driver does not support required encode feedback flags "
- "(BUFFER_OFFSET and BYTES_WRITTEN).\n");
- return AVERROR(ENOTSUP);
- }
-
ctx->base.op = &vulkan_base_encode_ops;
ctx->codec = codec;
@@ -881,6 +873,14 @@ av_cold int ff_vulkan_encode_init(AVCode
return AVERROR_EXTERNAL;
}
+ if ((ctx->enc_caps.supportedEncodeFeedbackFlags & feedback_flags) !=
+ feedback_flags) {
+ av_log(avctx, AV_LOG_ERROR,
+ "Driver does not support required encode feedback flags "
+ "(BUFFER_OFFSET and BYTES_WRITTEN).\n");
+ return AVERROR(ENOTSUP);
+ }
+
err = init_rc(avctx, ctx);
if (err < 0)
return err;
Index: FFmpeg/libavcodec/vulkan_encode.h
===================================================================
--- FFmpeg.orig/libavcodec/vulkan_encode.h
+++ FFmpeg/libavcodec/vulkan_encode.h
@@ -203,7 +203,7 @@ typedef struct FFVulkanEncodeContext {
#define VULKAN_ENCODE_COMMON_OPTIONS \
{ "qp", "Use an explicit constant quantizer for the whole stream", OFFSET(common.opts.qp), AV_OPT_TYPE_INT, { .i64 = -1 }, -1, 255, FLAGS }, \
- { "quality", "Set encode quality (trades off against speed, higher is faster)", OFFSET(common.opts.quality), AV_OPT_TYPE_INT, { .i64 = 0 }, 0, INT_MAX, FLAGS }, \
+ { "quality", "Set encode quality (trades off against speed, higher is slower)", OFFSET(common.opts.quality), AV_OPT_TYPE_INT, { .i64 = 0 }, 0, INT_MAX, FLAGS }, \
{ "rc_mode", "Select rate control type", OFFSET(common.opts.rc_mode), AV_OPT_TYPE_INT, { .i64 = FF_VK_RC_MODE_AUTO }, 0, FF_VK_RC_MODE_AUTO, FLAGS, "rc_mode" }, \
{ "auto", "Choose mode automatically based on parameters", 0, AV_OPT_TYPE_CONST, { .i64 = FF_VK_RC_MODE_AUTO }, INT_MIN, INT_MAX, FLAGS, "rc_mode" }, \
{ "driver", "Driver-specific rate control", 0, AV_OPT_TYPE_CONST, { .i64 = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DEFAULT_KHR }, INT_MIN, INT_MAX, FLAGS, "rc_mode" }, \
Index: FFmpeg/libavcodec/vulkan_encode_h265.c
===================================================================
--- FFmpeg.orig/libavcodec/vulkan_encode_h265.c
+++ FFmpeg/libavcodec/vulkan_encode_h265.c
@@ -1595,23 +1595,21 @@ static av_cold int vulkan_encode_h265_in
av_log(avctx, AV_LOG_VERBOSE, " Capability flags:\n");
av_log(avctx, AV_LOG_VERBOSE, " hdr_compliance: %i\n",
- !!(enc->caps.flags & VK_VIDEO_ENCODE_H264_CAPABILITY_HRD_COMPLIANCE_BIT_KHR));
+ !!(enc->caps.flags & VK_VIDEO_ENCODE_H265_CAPABILITY_HRD_COMPLIANCE_BIT_KHR));
av_log(avctx, AV_LOG_VERBOSE, " pred_weight_table_generated: %i\n",
- !!(enc->caps.flags & VK_VIDEO_ENCODE_H264_CAPABILITY_PREDICTION_WEIGHT_TABLE_GENERATED_BIT_KHR));
+ !!(enc->caps.flags & VK_VIDEO_ENCODE_H265_CAPABILITY_PREDICTION_WEIGHT_TABLE_GENERATED_BIT_KHR));
av_log(avctx, AV_LOG_VERBOSE, " row_unaligned_slice: %i\n",
- !!(enc->caps.flags & VK_VIDEO_ENCODE_H264_CAPABILITY_ROW_UNALIGNED_SLICE_BIT_KHR));
+ !!(enc->caps.flags & VK_VIDEO_ENCODE_H265_CAPABILITY_ROW_UNALIGNED_SLICE_SEGMENT_BIT_KHR));
av_log(avctx, AV_LOG_VERBOSE, " different_slice_type: %i\n",
- !!(enc->caps.flags & VK_VIDEO_ENCODE_H264_CAPABILITY_DIFFERENT_SLICE_TYPE_BIT_KHR));
+ !!(enc->caps.flags & VK_VIDEO_ENCODE_H265_CAPABILITY_DIFFERENT_SLICE_SEGMENT_TYPE_BIT_KHR));
av_log(avctx, AV_LOG_VERBOSE, " b_frame_in_l0_list: %i\n",
- !!(enc->caps.flags & VK_VIDEO_ENCODE_H264_CAPABILITY_B_FRAME_IN_L0_LIST_BIT_KHR));
+ !!(enc->caps.flags & VK_VIDEO_ENCODE_H265_CAPABILITY_B_FRAME_IN_L0_LIST_BIT_KHR));
av_log(avctx, AV_LOG_VERBOSE, " b_frame_in_l1_list: %i\n",
- !!(enc->caps.flags & VK_VIDEO_ENCODE_H264_CAPABILITY_B_FRAME_IN_L1_LIST_BIT_KHR));
+ !!(enc->caps.flags & VK_VIDEO_ENCODE_H265_CAPABILITY_B_FRAME_IN_L1_LIST_BIT_KHR));
av_log(avctx, AV_LOG_VERBOSE, " per_pict_type_min_max_qp: %i\n",
- !!(enc->caps.flags & VK_VIDEO_ENCODE_H264_CAPABILITY_PER_PICTURE_TYPE_MIN_MAX_QP_BIT_KHR));
+ !!(enc->caps.flags & VK_VIDEO_ENCODE_H265_CAPABILITY_PER_PICTURE_TYPE_MIN_MAX_QP_BIT_KHR));
av_log(avctx, AV_LOG_VERBOSE, " per_slice_constant_qp: %i\n",
- !!(enc->caps.flags & VK_VIDEO_ENCODE_H264_CAPABILITY_PER_SLICE_CONSTANT_QP_BIT_KHR));
- av_log(avctx, AV_LOG_VERBOSE, " generate_prefix_nalu: %i\n",
- !!(enc->caps.flags & VK_VIDEO_ENCODE_H264_CAPABILITY_GENERATE_PREFIX_NALU_BIT_KHR));
+ !!(enc->caps.flags & VK_VIDEO_ENCODE_H265_CAPABILITY_PER_SLICE_SEGMENT_CONSTANT_QP_BIT_KHR));
av_log(avctx, AV_LOG_VERBOSE, " Capabilities:\n");
av_log(avctx, AV_LOG_VERBOSE, " maxLevelIdc: %i\n",
Index: FFmpeg/libavutil/hwcontext_vulkan.c
===================================================================
--- FFmpeg.orig/libavutil/hwcontext_vulkan.c
+++ FFmpeg/libavutil/hwcontext_vulkan.c
@@ -190,11 +190,18 @@ typedef struct VulkanFramesPriv {
/* Properties for DRM modifier for each plane in the image */
VkDrmFormatModifierPropertiesEXT drm_format_modifier_properties[5];
+
+ /* Set when physical device reports DEDICATED_ONLY for DMA-BUF export (try_export_flags) */
+ int export_requires_dedicated;
} VulkanFramesPriv;
typedef struct AVVkFrameInternal {
pthread_mutex_t update_mutex;
+ /* Binary semaphore for SYNC_FD export at DRM map time. Created once lazily,
+ * re-signaled each time via a submit in vulkan_map_to_drm. */
+ VkSemaphore drm_sync_sem;
+
#if CONFIG_CUDA
/* Importing external memory into cuda is really expensive so we keep the
* memory imported all the time */
@@ -2370,7 +2377,7 @@ static int alloc_mem(AVHWDeviceContext *
return 0;
}
-static void vulkan_free_internal(AVVkFrame *f)
+static void vulkan_free_internal(VulkanDevicePriv *p, AVVkFrame *f)
{
av_unused AVVkFrameInternal *internal = f->internal;
@@ -2402,6 +2409,10 @@ static void vulkan_free_internal(AVVkFra
}
#endif
+ if (internal->drm_sync_sem != VK_NULL_HANDLE)
+ p->vkctx.vkfn.DestroySemaphore(p->p.act_dev, internal->drm_sync_sem,
+ p->p.alloc);
+
pthread_mutex_destroy(&internal->update_mutex);
av_freep(&f->internal);
}
@@ -2429,7 +2440,7 @@ static void vulkan_frame_free(AVHWFrames
vk->WaitSemaphores(hwctx->act_dev, &sem_wait, UINT64_MAX);
}
- vulkan_free_internal(f);
+ vulkan_free_internal(p, f);
for (int i = 0; i < nb_images; i++) {
vk->DestroyImage(hwctx->act_dev, f->img[i], hwctx->alloc);
@@ -2476,6 +2487,10 @@ static int alloc_bind_mem(AVHWFramesCont
vk->GetImageMemoryRequirements2(hwctx->act_dev, &req_desc, &req);
+ av_log(hwfc, AV_LOG_TRACE,
+ "plane %d: driver reports prefersDedicatedAllocation=%i requiresDedicatedAllocation=%i\n",
+ img_cnt, ded_req.prefersDedicatedAllocation, ded_req.requiresDedicatedAllocation);
+
if (f->tiling == VK_IMAGE_TILING_LINEAR)
req.memoryRequirements.size = FFALIGN(req.memoryRequirements.size,
p->props.properties.limits.minMemoryMapAlignment);
@@ -2483,6 +2498,8 @@ static int alloc_bind_mem(AVHWFramesCont
/* In case the implementation prefers/requires dedicated allocation */
use_ded_mem = ded_req.prefersDedicatedAllocation |
ded_req.requiresDedicatedAllocation;
+ if (((VulkanFramesPriv *)hwfc->hwctx)->export_requires_dedicated)
+ use_ded_mem = 1;
if (use_ded_mem)
ded_alloc.image = f->img[img_cnt];
@@ -2854,7 +2871,8 @@ static void try_export_flags(AVHWFramesC
.type = VK_IMAGE_TYPE_2D,
.tiling = hwctx->tiling,
.usage = hwctx->usage,
- .flags = VK_IMAGE_CREATE_ALIAS_BIT,
+ .flags = (hwctx->tiling == VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT && has_mods) ?
+ (hwctx->img_flags) : (VkImageCreateFlags)(VK_IMAGE_CREATE_ALIAS_BIT),
};
nb_mods = has_mods ? drm_mod_info->drmFormatModifierCount : 1;
@@ -2865,9 +2883,18 @@ static void try_export_flags(AVHWFramesC
ret = vk->GetPhysicalDeviceImageFormatProperties2(dev_hwctx->phys_dev,
&pinfo, &props);
+ if (has_mods)
+ av_log(hwfc, AV_LOG_VERBOSE, "GetPhysicalDeviceImageFormatProperties2: mod[%d]=0x%llx -> %s\n",
+ i, (unsigned long long)phy_dev_mod_info.drmFormatModifier,
+ ret == VK_SUCCESS ? "OK" : "FAIL");
if (ret == VK_SUCCESS) {
*iexp |= exp;
*comp_handle_types |= eprops.externalMemoryProperties.compatibleHandleTypes;
+ if (exp == VK_EXTERNAL_MEMORY_HANDLE_TYPE_DMA_BUF_BIT_EXT) {
+ VulkanFramesPriv *fp = hwfc->hwctx;
+ fp->export_requires_dedicated = !!(eprops.externalMemoryProperties.externalMemoryFeatures &
+ VK_EXTERNAL_MEMORY_FEATURE_DEDICATED_ONLY_BIT);
+ }
}
}
}
@@ -2895,7 +2922,8 @@ static AVBufferRef *vulkan_pool_alloc(vo
? VK_EXTERNAL_MEMORY_HANDLE_TYPE_OPAQUE_WIN32_BIT
: VK_EXTERNAL_MEMORY_HANDLE_TYPE_OPAQUE_WIN32_KMT_BIT);
#else
- if (p->vkctx.extensions & FF_VK_EXT_EXTERNAL_FD_MEMORY)
+ if ((p->vkctx.extensions & FF_VK_EXT_EXTERNAL_FD_MEMORY) &&
+ (hwctx->tiling != VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT))
try_export_flags(hwfc, &eiinfo.handleTypes, &e,
VK_EXTERNAL_MEMORY_HANDLE_TYPE_OPAQUE_FD_BIT);
@@ -2914,8 +2942,10 @@ static AVBufferRef *vulkan_pool_alloc(vo
err = create_frame(hwfc, &f, hwctx->tiling, hwctx->usage, hwctx->img_flags,
hwctx->nb_layers,
eiinfo.handleTypes ? &eiinfo : hwctx->create_pnext);
- if (err)
+ if (err) {
+ av_log(hwfc, AV_LOG_ERROR, "vulkan_pool_alloc failed: create_frame failed: %d\n", err);
return NULL;
+ }
err = alloc_bind_mem(hwfc, f, eminfo, sizeof(*eminfo));
if (err)
@@ -2943,6 +2973,7 @@ static AVBufferRef *vulkan_pool_alloc(vo
return avbuf;
fail:
+ av_log(hwfc, AV_LOG_ERROR, "vulkan_pool_alloc failed with error %d\n", err);
vulkan_frame_free(hwfc, f);
return NULL;
}
@@ -3044,7 +3075,14 @@ static int vulkan_frames_init(AVHWFrames
}
/* Lone DPB images do not need additional flags. */
- if (!is_lone_dpb) {
+ /* With DRM modifier + video profile the caller has already chosen a valid
+ * usage/img_flags/chain; do not add usage or img_flags (supported_usage does
+ * not consider the actual modifier or video profile). */
+ int drm_mod_with_video = (hwctx->tiling == VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT &&
+ ff_vk_find_struct(hwctx->create_pnext,
+ VK_STRUCTURE_TYPE_VIDEO_PROFILE_LIST_INFO_KHR));
+
+ if (!is_lone_dpb && !drm_mod_with_video) {
/* Image usage flags */
hwctx->usage |= supported_usage & (VK_IMAGE_USAGE_TRANSFER_DST_BIT |
VK_IMAGE_USAGE_TRANSFER_SRC_BIT |
@@ -3163,8 +3201,8 @@ static int vulkan_frames_init(AVHWFrames
}
vk->GetPhysicalDeviceFormatProperties2(dev_hwctx->phys_dev, fmt->fallback[i], &fmtp);
- for (uint32_t i = 0; i < modp.drmFormatModifierCount; ++i) {
- VkDrmFormatModifierPropertiesEXT *m = &modp.pDrmFormatModifierProperties[i];
+ for (uint32_t j = 0; j < modp.drmFormatModifierCount; ++j) {
+ VkDrmFormatModifierPropertiesEXT *m = &modp.pDrmFormatModifierProperties[j];
if (m->drmFormatModifier == drm_mod.drmFormatModifier) {
mod_props = m;
break;
@@ -3502,7 +3540,7 @@ static int vulkan_map_from_drm_frame_des
&f->flags, &f->mem[i]);
if (err) {
close(idesc.fd);
- return err;
+ goto fail;
}
f->size[i] = req2.memoryRequirements.size;
@@ -3969,7 +4007,7 @@ static int vulkan_export_to_cuda(AVHWFra
return 0;
fail:
- vulkan_free_internal(dst_f);
+ vulkan_free_internal(p, dst_f);
return err;
}
@@ -3980,6 +4018,7 @@ static int vulkan_transfer_data_from_cud
CUcontext dummy;
AVVkFrame *dst_f;
AVVkFrameInternal *dst_int;
+ VulkanDevicePriv *p = hwfc->device_ctx->hwctx;
VulkanFramesPriv *fp = hwfc->hwctx;
const int planes = av_pix_fmt_count_planes(hwfc->sw_format);
const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(hwfc->sw_format);
@@ -4058,7 +4097,7 @@ static int vulkan_transfer_data_from_cud
fail:
CHECK_CU(cu->cuCtxPopCurrent(&dummy));
- vulkan_free_internal(dst_f);
+ vulkan_free_internal(p, dst_f);
av_buffer_unref(&dst->buf[0]);
return err;
}
@@ -4128,6 +4167,72 @@ static VkImageAspectFlags plane_index_to
return VK_IMAGE_ASPECT_MEMORY_PLANE_0_BIT_EXT;
}
+#ifdef DMA_BUF_IOCTL_EXPORT_SYNC_FILE
+static int vulkan_drm_export_sync_fd(AVHWFramesContext *hwfc, AVVkFrame *f,
+ VulkanFramesPriv *fp, int nb_sems)
+{
+ int sync_fd = -1;
+ VkResult ret;
+ VulkanDevicePriv *p = hwfc->device_ctx->hwctx;
+ AVVulkanDeviceContext *hwctx = &p->p;
+ FFVulkanFunctions *vk = &p->vkctx.vkfn;
+
+ if (f->internal->drm_sync_sem == VK_NULL_HANDLE) {
+ VkExportSemaphoreCreateInfo exp_info = {
+ .sType = VK_STRUCTURE_TYPE_EXPORT_SEMAPHORE_CREATE_INFO,
+ .handleTypes = VK_EXTERNAL_SEMAPHORE_HANDLE_TYPE_SYNC_FD_BIT,
+ };
+ VkSemaphoreTypeCreateInfo type_info = {
+ .sType = VK_STRUCTURE_TYPE_SEMAPHORE_TYPE_CREATE_INFO,
+ .pNext = &exp_info,
+ .semaphoreType = VK_SEMAPHORE_TYPE_BINARY,
+ };
+ VkSemaphoreCreateInfo sem_create = {
+ .sType = VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO,
+ .pNext = &type_info,
+ };
+ ret = vk->CreateSemaphore(hwctx->act_dev, &sem_create, hwctx->alloc,
+ &f->internal->drm_sync_sem);
+ if (ret != VK_SUCCESS) {
+ av_log(hwctx, AV_LOG_ERROR, "Failed to create DRM export semaphore: %s\n",
+ ff_vk_ret2str(ret));
+ return AVERROR_EXTERNAL;
+ }
+ }
+
+ /* Submit a lightweight exec that waits on the timeline semaphore
+ * (true last operation on the frame) and signals the binary semaphore,
+ * so any Vulkan frame can get a SYNC_FD regardless of origin. */
+ FFVkExecContext *exec = ff_vk_exec_get(&p->vkctx, &fp->compute_exec);
+ if (ff_vk_exec_start(&p->vkctx, exec) >= 0) {
+ for (int i = 0; i < nb_sems; i++)
+ ff_vk_exec_add_dep_wait_sem(&p->vkctx, exec, f->sem[i],
+ f->sem_value[i],
+ VK_PIPELINE_STAGE_2_ALL_COMMANDS_BIT);
+ ff_vk_exec_add_dep_bool_sem(&p->vkctx, exec, &f->internal->drm_sync_sem, 1,
+ VK_PIPELINE_STAGE_2_ALL_COMMANDS_BIT, 0);
+ if (ff_vk_exec_submit(&p->vkctx, exec) >= 0) {
+ VkSemaphoreGetFdInfoKHR get_fd_info = {
+ .sType = VK_STRUCTURE_TYPE_SEMAPHORE_GET_FD_INFO_KHR,
+ .semaphore = f->internal->drm_sync_sem,
+ .handleType = VK_EXTERNAL_SEMAPHORE_HANDLE_TYPE_SYNC_FD_BIT,
+ };
+ ret = vk->GetSemaphoreFdKHR(hwctx->act_dev, &get_fd_info, &sync_fd);
+ if (ret != VK_SUCCESS) {
+ av_log(hwctx, AV_LOG_WARNING,
+ "Failed to get sync fd from DRM map export semaphore: %s\n",
+ ff_vk_ret2str(ret));
+ sync_fd = -1;
+ }
+ } else {
+ ff_vk_exec_discard_deps(&p->vkctx, exec);
+ }
+ }
+
+ return sync_fd;
+}
+#endif
+
static int vulkan_map_to_drm(AVHWFramesContext *hwfc, AVFrame *dst,
const AVFrame *src, int flags)
{
@@ -4138,15 +4243,14 @@ static int vulkan_map_to_drm(AVHWFramesC
AVVulkanDeviceContext *hwctx = &p->p;
FFVulkanFunctions *vk = &p->vkctx.vkfn;
VulkanFramesPriv *fp = hwfc->hwctx;
+ const int planes = av_pix_fmt_count_planes(hwfc->sw_format);
const int nb_images = ff_vk_count_images(f);
VkImageDrmFormatModifierPropertiesEXT drm_mod = {
.sType = VK_STRUCTURE_TYPE_IMAGE_DRM_FORMAT_MODIFIER_PROPERTIES_EXT,
};
- VkSemaphoreWaitInfo wait_info = {
- .sType = VK_STRUCTURE_TYPE_SEMAPHORE_WAIT_INFO,
- .flags = 0x0,
- .semaphoreCount = nb_images,
- };
+ const int nb_sems = nb_images;
+ int free_drm_desc_on_err = 1;
+ int sync_fd = -1;
AVDRMFrameDescriptor *drm_desc = av_mallocz(sizeof(*drm_desc));
if (!drm_desc)
@@ -4156,16 +4260,36 @@ static int vulkan_map_to_drm(AVHWFramesC
if (err < 0)
goto end;
- /* Wait for the operation to finish so we can cleanly export it. */
- wait_info.pSemaphores = f->sem;
- wait_info.pValues = f->sem_value;
+#ifdef DMA_BUF_IOCTL_EXPORT_SYNC_FILE
+ if ((p->vkctx.extensions & FF_VK_EXT_EXTERNAL_FD_SEM) &&
+ f->tiling == VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT &&
+ vk->GetSemaphoreFdKHR && vk->CreateSemaphore) {
+ err = vulkan_drm_export_sync_fd(hwfc, f, fp, nb_sems);
+ if (err < 0)
+ goto end;
+ sync_fd = err;
+ err = 0;
+ }
+#endif
- vk->WaitSemaphores(hwctx->act_dev, &wait_info, UINT64_MAX);
+ if (sync_fd < 0) {
+ VkSemaphoreWaitInfo wait_info = {
+ .sType = VK_STRUCTURE_TYPE_SEMAPHORE_WAIT_INFO,
+ .flags = 0x0,
+ .semaphoreCount = nb_sems,
+ .pSemaphores = f->sem,
+ .pValues = f->sem_value,
+ };
+ vk->WaitSemaphores(hwctx->act_dev, &wait_info, UINT64_MAX);
+ }
err = ff_hwframe_map_create(src->hw_frames_ctx, dst, src, &vulkan_unmap_to_drm, drm_desc);
if (err < 0)
goto end;
+ /* It will be freed in ff_hwframe_map_create callback */
+ free_drm_desc_on_err = 0;
+
ret = vk->GetImageDrmFormatModifierPropertiesEXT(hwctx->act_dev, f->img[0],
&drm_mod);
if (ret != VK_SUCCESS) {
@@ -4174,7 +4298,7 @@ static int vulkan_map_to_drm(AVHWFramesC
goto end;
}
- for (int i = 0; (i < nb_images) && (f->mem[i]); i++) {
+ for (int i = 0; (i < planes) && (f->mem[i]); i++) {
VkMemoryGetFdInfoKHR export_info = {
.sType = VK_STRUCTURE_TYPE_MEMORY_GET_FD_INFO_KHR,
.memory = f->mem[i],
@@ -4189,12 +4313,30 @@ static int vulkan_map_to_drm(AVHWFramesC
goto end;
}
+#if HAVE_LINUX_DMA_BUF_H && defined(DMA_BUF_IOCTL_IMPORT_SYNC_FILE)
+ if (sync_fd >= 0) {
+ int dup_fd = dup(sync_fd);
+ if (dup_fd >= 0) {
+ struct dma_buf_import_sync_file import_info = {
+ .flags = DMA_BUF_SYNC_WRITE,
+ .fd = dup_fd,
+ };
+ if (ioctl(drm_desc->objects[i].fd, DMA_BUF_IOCTL_IMPORT_SYNC_FILE, &import_info) < 0)
+ av_log(hwfc, AV_LOG_WARNING, "DMA_BUF_IOCTL_IMPORT_SYNC_FILE failed: %s\n", av_err2str(AVERROR(errno)));
+ close(dup_fd);
+ } else {
+ av_log(hwfc, AV_LOG_WARNING, "dup(sync_fd) failed: %s\n", av_err2str(AVERROR(errno)));
+ }
+ }
+#endif
+
drm_desc->nb_objects++;
drm_desc->objects[i].size = f->size[i];
drm_desc->objects[i].format_modifier = drm_mod.drmFormatModifier;
}
- drm_desc->nb_layers = nb_images;
+ /* NV12 has 2 planes but 1 image/semaphore */
+ drm_desc->nb_layers = FFMAX(planes, nb_images);
for (int i = 0; i < drm_desc->nb_layers; i++) {
VkFormat plane_vkfmt = av_vkfmt_from_pixfmt(hwfc->sw_format)[i];
@@ -4209,13 +4351,14 @@ static int vulkan_map_to_drm(AVHWFramesC
for (int j = 0; j < drm_desc->layers[i].nb_planes; j++) {
VkSubresourceLayout layout;
+ int aspect_plane = (nb_images == 1) ? i : j;
VkImageSubresource sub = {
- .aspectMask = plane_index_to_aspect(j),
+ .aspectMask = plane_index_to_aspect(aspect_plane),
};
drm_desc->layers[i].planes[j].object_index = FFMIN(i, drm_desc->nb_objects - 1);
- vk->GetImageSubresourceLayout(hwctx->act_dev, f->img[i], &sub, &layout);
+ vk->GetImageSubresourceLayout(hwctx->act_dev, f->img[FFMIN(i, nb_images - 1)], &sub, &layout);
drm_desc->layers[i].planes[j].offset = layout.offset;
drm_desc->layers[i].planes[j].pitch = layout.rowPitch;
}
@@ -4235,13 +4378,22 @@ static int vulkan_map_to_drm(AVHWFramesC
dst->width = src->width;
dst->height = src->height;
dst->data[0] = (uint8_t *)drm_desc;
+ dst->hw_frames_ctx = av_buffer_ref(src->hw_frames_ctx);
+
+ if (sync_fd >= 0)
+ close(sync_fd);
av_log(hwfc, AV_LOG_VERBOSE, "Mapped AVVkFrame to a DRM object!\n");
return 0;
end:
- av_free(drm_desc);
+ for (int i = 0; i < drm_desc->nb_objects; i++)
+ close(drm_desc->objects[i].fd);
+ if (free_drm_desc_on_err)
+ av_free(drm_desc);
+ if (sync_fd >= 0)
+ close(sync_fd);
return err;
}
@@ -4751,6 +4903,7 @@ static int vulkan_transfer_data_to_cuda(
CUcontext dummy;
AVVkFrame *dst_f;
AVVkFrameInternal *dst_int;
+ VulkanDevicePriv *p = hwfc->device_ctx->hwctx;
VulkanFramesPriv *fp = hwfc->hwctx;
const int planes = av_pix_fmt_count_planes(hwfc->sw_format);
const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(hwfc->sw_format);
@@ -4831,7 +4984,7 @@ static int vulkan_transfer_data_to_cuda(
fail:
CHECK_CU(cu->cuCtxPopCurrent(&dummy));
- vulkan_free_internal(dst_f);
+ vulkan_free_internal(p, dst_f);
av_buffer_unref(&dst->buf[0]);
return err;
}
@@ -1,58 +0,0 @@
Index: FFmpeg/libavutil/hwcontext_vaapi.c
===================================================================
--- FFmpeg.orig/libavutil/hwcontext_vaapi.c
+++ FFmpeg/libavutil/hwcontext_vaapi.c
@@ -72,6 +72,7 @@ typedef struct VAAPIDevicePriv {
typedef struct VAAPISurfaceFormat {
enum AVPixelFormat pix_fmt;
VAImageFormat image_format;
+ unsigned int fourcc;
} VAAPISurfaceFormat;
typedef struct VAAPIDeviceContext {
@@ -221,15 +222,21 @@ static int vaapi_get_image_format(AVHWDe
VAImageFormat **image_format)
{
VAAPIDeviceContext *ctx = hwdev->hwctx;
+ const VAAPIFormatDescriptor *desc;
int i;
+ desc = vaapi_format_from_pix_fmt(pix_fmt);
+ if (!desc || !image_format)
+ goto fail;
+
for (i = 0; i < ctx->nb_formats; i++) {
- if (ctx->formats[i].pix_fmt == pix_fmt) {
- if (image_format)
- *image_format = &ctx->formats[i].image_format;
+ if (ctx->formats[i].fourcc == desc->fourcc) {
+ *image_format = &ctx->formats[i].image_format;
return 0;
}
}
+
+fail:
return AVERROR(ENOSYS);
}
@@ -438,6 +445,7 @@ static int vaapi_device_init(AVHWDeviceC
av_log(hwdev, AV_LOG_DEBUG, "Format %#x -> %s.\n",
fourcc, av_get_pix_fmt_name(pix_fmt));
ctx->formats[ctx->nb_formats].pix_fmt = pix_fmt;
+ ctx->formats[ctx->nb_formats].fourcc = fourcc;
ctx->formats[ctx->nb_formats].image_format = image_list[i];
++ctx->nb_formats;
}
@@ -1014,12 +1022,6 @@ static int vaapi_map_to_memory(AVHWFrame
{
int err;
- if (dst->format != AV_PIX_FMT_NONE) {
- err = vaapi_get_image_format(hwfc->device_ctx, dst->format, NULL);
- if (err < 0)
- return err;
- }
-
err = vaapi_map_frame(hwfc, dst, src, flags);
if (err)
return err;
@@ -0,0 +1,769 @@
Index: FFmpeg/libavutil/hwcontext_vulkan.c
===================================================================
--- FFmpeg.orig/libavutil/hwcontext_vulkan.c
+++ FFmpeg/libavutil/hwcontext_vulkan.c
@@ -59,11 +59,47 @@
#include <xf86drm.h>
#include <drm_fourcc.h>
#include "hwcontext_drm.h"
+#ifndef DRM_FORMAT_P010
+#define DRM_FORMAT_P010 fourcc_code('P', '0', '1', '0')
+#define DRM_FORMAT_P012 fourcc_code('P', '0', '1', '2')
+#define DRM_FORMAT_P016 fourcc_code('P', '0', '1', '6')
+#endif
+#ifndef DRM_FORMAT_P210
+#define DRM_FORMAT_P210 fourcc_code('P', '2', '1', '0')
+#endif
+#ifndef DRM_FORMAT_S010
+#define DRM_FORMAT_S010 fourcc_code('S', '0', '1', '0')
+#define DRM_FORMAT_S210 fourcc_code('S', '2', '1', '0')
+#define DRM_FORMAT_S410 fourcc_code('S', '4', '1', '0')
+#define DRM_FORMAT_S012 fourcc_code('S', '0', '1', '2')
+#define DRM_FORMAT_S212 fourcc_code('S', '2', '1', '2')
+#define DRM_FORMAT_S412 fourcc_code('S', '4', '1', '2')
+#define DRM_FORMAT_S016 fourcc_code('S', '0', '1', '6')
+#define DRM_FORMAT_S216 fourcc_code('S', '2', '1', '6')
+#define DRM_FORMAT_S416 fourcc_code('S', '4', '1', '6')
+#endif
#endif
#if HAVE_LINUX_DMA_BUF_H
#include <sys/ioctl.h>
#include <linux/dma-buf.h>
+/* dma-buf export/import sync is only available in v6.0+ */
+#ifndef DMA_BUF_IOCTL_EXPORT_SYNC_FILE
+struct dma_buf_export_sync_file {
+ uint32_t flags;
+ int32_t fd;
+};
+#define DMA_BUF_IOCTL_EXPORT_SYNC_FILE \
+ _IOWR(DMA_BUF_BASE, 2, struct dma_buf_export_sync_file)
+#endif
+#ifndef DMA_BUF_IOCTL_IMPORT_SYNC_FILE
+struct dma_buf_import_sync_file {
+ uint32_t flags;
+ int32_t fd;
+};
+#define DMA_BUF_IOCTL_IMPORT_SYNC_FILE \
+ _IOW(DMA_BUF_BASE, 3, struct dma_buf_import_sync_file)
+#endif
#endif
#if CONFIG_CUDA
@@ -167,6 +203,11 @@ typedef struct VulkanDevicePriv {
/* Maximum queues */
int limit_queues;
+
+ /* DRM dma-buf sync file ioctl ret code, tested on first use.
+ * 0: success, other: fail (use fallback) */
+ int drm_export_sync_file_ret;
+ int drm_import_sync_file_ret;
} VulkanDevicePriv;
typedef struct VulkanFramesPriv {
@@ -1794,6 +1835,7 @@ static void vulkan_device_uninit(AVHWDev
static int vulkan_device_create_internal(AVHWDeviceContext *ctx,
VulkanDeviceSelection *dev_select,
+ int use_linear_images,
int disable_multiplane,
AVDictionary *opts, int flags)
{
@@ -1866,10 +1908,14 @@ static int vulkan_device_create_internal
goto end;
}
- /* Tiled images setting, use them by default */
- opt_d = av_dict_get(opts, "linear_images", NULL, 0);
- if (opt_d)
- p->use_linear_images = strtol(opt_d->value, NULL, 10);
+ /* Tiled images setting, use them by default.
+ * The use_linear_images argument takes precedent over the option */
+ p->use_linear_images = use_linear_images;
+ if (!p->use_linear_images) {
+ opt_d = av_dict_get(opts, "linear_images", NULL, 0);
+ if (opt_d)
+ p->use_linear_images = strtol(opt_d->value, NULL, 10);
+ }
/* The disable_multiplane argument takes precedent over the option */
p->disable_multiplane = disable_multiplane;
@@ -2167,7 +2213,7 @@ static int vulkan_device_create(AVHWDevi
}
}
- return vulkan_device_create_internal(ctx, &dev_select, 0, opts, flags);
+ return vulkan_device_create_internal(ctx, &dev_select, 0, 0, opts, flags);
}
static int vulkan_device_derive(AVHWDeviceContext *ctx,
@@ -2209,7 +2255,8 @@ static int vulkan_device_derive(AVHWDevi
dev_select.vendor_id = 0x1002;
}
- return vulkan_device_create_internal(ctx, &dev_select, 0, opts, flags);
+ /* Linear multiplane image is used to ensure compatibility with VAAPI */
+ return vulkan_device_create_internal(ctx, &dev_select, 1, 0, opts, flags);
}
#endif
#if CONFIG_LIBDRM
@@ -2242,7 +2289,14 @@ static int vulkan_device_derive(AVHWDevi
drmFreeDevice(&drm_dev_info);
- return vulkan_device_create_internal(ctx, &dev_select, 0, opts, flags);
+ /* Linear multiplane image is used to ensure compatibility with DRM */
+ return vulkan_device_create_internal(ctx, &dev_select, 1, 0, opts, flags);
+ }
+#endif
+#if CONFIG_RKMPP
+ case AV_HWDEVICE_TYPE_RKMPP: {
+ /* Linear multiplane image is used to ensure compatibility with RKMPP */
+ return vulkan_device_create_internal(ctx, &dev_select, 1, 0, opts, flags);
}
#endif
#if CONFIG_CUDA
@@ -2265,7 +2319,7 @@ static int vulkan_device_derive(AVHWDevi
* CUDA is not able to import multiplane images, so always derive a
* Vulkan device with multiplane disabled.
*/
- return vulkan_device_create_internal(ctx, &dev_select, 1, opts, flags);
+ return vulkan_device_create_internal(ctx, &dev_select, 0, 1, opts, flags);
}
#endif
default:
@@ -2874,6 +2928,8 @@ static void try_export_flags(AVHWFramesC
.flags = (hwctx->tiling == VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT && has_mods) ?
(hwctx->img_flags) : (VkImageCreateFlags)(VK_IMAGE_CREATE_ALIAS_BIT),
};
+ VkPhysicalDeviceImageFormatInfo2 pinfo_fallback = pinfo;
+ pinfo_fallback.format = av_vkfmt_from_pixfmt(hwfc->sw_format)[0];
nb_mods = has_mods ? drm_mod_info->drmFormatModifierCount : 1;
for (int i = 0; i < nb_mods; i++) {
@@ -2881,7 +2937,10 @@ static void try_export_flags(AVHWFramesC
phy_dev_mod_info.drmFormatModifier = drm_mod_info->pDrmFormatModifiers[i];
ret = vk->GetPhysicalDeviceImageFormatProperties2(dev_hwctx->phys_dev,
- &pinfo, &props);
+ &pinfo, &props);
+ if (ret != VK_SUCCESS)
+ ret = vk->GetPhysicalDeviceImageFormatProperties2(dev_hwctx->phys_dev,
+ &pinfo_fallback, &props);
if (has_mods)
av_log(hwfc, AV_LOG_VERBOSE, "GetPhysicalDeviceImageFormatProperties2: mod[%d]=0x%llx -> %s\n",
@@ -3285,28 +3344,65 @@ static void vulkan_unmap_from_drm(AVHWFr
static const struct {
uint32_t drm_fourcc;
VkFormat vk_format;
+ int nb_planes;
+ uint32_t fallbacks[AV_DRM_MAX_PLANES];
} vulkan_drm_format_map[] = {
- { DRM_FORMAT_R8, VK_FORMAT_R8_UNORM },
- { DRM_FORMAT_R16, VK_FORMAT_R16_UNORM },
- { DRM_FORMAT_GR88, VK_FORMAT_R8G8_UNORM },
- { DRM_FORMAT_RG88, VK_FORMAT_R8G8_UNORM },
- { DRM_FORMAT_GR1616, VK_FORMAT_R16G16_UNORM },
- { DRM_FORMAT_RG1616, VK_FORMAT_R16G16_UNORM },
- { DRM_FORMAT_ARGB8888, VK_FORMAT_B8G8R8A8_UNORM },
- { DRM_FORMAT_XRGB8888, VK_FORMAT_B8G8R8A8_UNORM },
- { DRM_FORMAT_ABGR8888, VK_FORMAT_R8G8B8A8_UNORM },
- { DRM_FORMAT_XBGR8888, VK_FORMAT_R8G8B8A8_UNORM },
- { DRM_FORMAT_ARGB2101010, VK_FORMAT_A2B10G10R10_UNORM_PACK32 },
- { DRM_FORMAT_ABGR2101010, VK_FORMAT_A2R10G10B10_UNORM_PACK32 },
- { DRM_FORMAT_XRGB2101010, VK_FORMAT_A2B10G10R10_UNORM_PACK32 },
- { DRM_FORMAT_XBGR2101010, VK_FORMAT_A2R10G10B10_UNORM_PACK32 },
+ { DRM_FORMAT_R8, VK_FORMAT_R8_UNORM, 1, { DRM_FORMAT_R8 } },
+ { DRM_FORMAT_R16, VK_FORMAT_R16_UNORM, 1, { DRM_FORMAT_R16 } },
+ { DRM_FORMAT_GR88, VK_FORMAT_R8G8_UNORM, 1, { DRM_FORMAT_GR88 } },
+ { DRM_FORMAT_RG88, VK_FORMAT_R8G8_UNORM, 1, { DRM_FORMAT_RG88 } },
+ { DRM_FORMAT_GR1616, VK_FORMAT_R16G16_UNORM, 1, { DRM_FORMAT_GR1616 } },
+ { DRM_FORMAT_RG1616, VK_FORMAT_R16G16_UNORM, 1, { DRM_FORMAT_RG1616 } },
+ { DRM_FORMAT_ARGB8888, VK_FORMAT_B8G8R8A8_UNORM, 1, { DRM_FORMAT_ARGB8888 } },
+ { DRM_FORMAT_XRGB8888, VK_FORMAT_B8G8R8A8_UNORM, 1, { DRM_FORMAT_XRGB8888 } },
+ { DRM_FORMAT_ABGR8888, VK_FORMAT_R8G8B8A8_UNORM, 1, { DRM_FORMAT_ABGR8888 } },
+ { DRM_FORMAT_XBGR8888, VK_FORMAT_R8G8B8A8_UNORM, 1, { DRM_FORMAT_XBGR8888 } },
+ { DRM_FORMAT_ARGB2101010, VK_FORMAT_A2B10G10R10_UNORM_PACK32, 1, { DRM_FORMAT_ARGB2101010 } },
+ { DRM_FORMAT_ABGR2101010, VK_FORMAT_A2R10G10B10_UNORM_PACK32, 1, { DRM_FORMAT_ABGR2101010 } },
+ { DRM_FORMAT_XRGB2101010, VK_FORMAT_A2B10G10R10_UNORM_PACK32, 1, { DRM_FORMAT_XRGB2101010 } },
+ { DRM_FORMAT_XBGR2101010, VK_FORMAT_A2R10G10B10_UNORM_PACK32, 1, { DRM_FORMAT_XBGR2101010 } },
// All these DRM_FORMATs were added in the same libdrm commit.
#ifdef DRM_FORMAT_XYUV8888
- { DRM_FORMAT_XYUV8888, VK_FORMAT_R8G8B8A8_UNORM },
- { DRM_FORMAT_XVYU2101010, VK_FORMAT_A2R10G10B10_UNORM_PACK32 } ,
- { DRM_FORMAT_XVYU12_16161616, VK_FORMAT_R12X4G12X4B12X4A12X4_UNORM_4PACK16 } ,
- { DRM_FORMAT_XVYU16161616, VK_FORMAT_R16G16B16A16_UNORM } ,
+ { DRM_FORMAT_XYUV8888, VK_FORMAT_R8G8B8A8_UNORM, 1, { DRM_FORMAT_XYUV8888 } },
+ { DRM_FORMAT_XVYU2101010, VK_FORMAT_A2R10G10B10_UNORM_PACK32, 1, { DRM_FORMAT_XVYU2101010 } },
+ { DRM_FORMAT_XVYU12_16161616, VK_FORMAT_R12X4G12X4B12X4A12X4_UNORM_4PACK16, 1, { DRM_FORMAT_XVYU12_16161616 } },
+ { DRM_FORMAT_XVYU16161616, VK_FORMAT_R16G16B16A16_UNORM, 1, { DRM_FORMAT_XVYU16161616 } },
+#endif
+
+ { DRM_FORMAT_RGB888, VK_FORMAT_R8G8B8_UNORM, 1, { DRM_FORMAT_RGB888 } },
+ { DRM_FORMAT_BGR888, VK_FORMAT_B8G8R8_UNORM, 1, { DRM_FORMAT_BGR888 } },
+
+ { DRM_FORMAT_NV12, VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, 2, { DRM_FORMAT_R8, DRM_FORMAT_GR88 } },
+ { DRM_FORMAT_NV16, VK_FORMAT_G8_B8R8_2PLANE_422_UNORM, 2, { DRM_FORMAT_R8, DRM_FORMAT_GR88 } },
+ { DRM_FORMAT_NV24, VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, 2, { DRM_FORMAT_R8, DRM_FORMAT_GR88 } },
+
+ // All these DRM_FORMATs were added in the same libdrm commit.
+#ifdef DRM_FORMAT_P010
+ { DRM_FORMAT_P010, VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, 2, { DRM_FORMAT_R16, DRM_FORMAT_GR1616 } },
+ { DRM_FORMAT_P012, VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16, 2, { DRM_FORMAT_R16, DRM_FORMAT_GR1616 } },
+ { DRM_FORMAT_P016, VK_FORMAT_G16_B16R16_2PLANE_420_UNORM, 2, { DRM_FORMAT_R16, DRM_FORMAT_GR1616 } },
+#endif
+
+#ifdef DRM_FORMAT_P210
+ { DRM_FORMAT_P210, VK_FORMAT_G10X6_B10X6R10X6_2PLANE_422_UNORM_3PACK16, 2, { DRM_FORMAT_R16, DRM_FORMAT_GR1616 } },
+#endif
+
+ { DRM_FORMAT_YUV420, VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM, 3, { DRM_FORMAT_R8, DRM_FORMAT_R8, DRM_FORMAT_R8 } },
+ { DRM_FORMAT_YUV422, VK_FORMAT_G8_B8_R8_3PLANE_422_UNORM, 3, { DRM_FORMAT_R8, DRM_FORMAT_R8, DRM_FORMAT_R8 } },
+ { DRM_FORMAT_YUV444, VK_FORMAT_G8_B8_R8_3PLANE_444_UNORM, 3, { DRM_FORMAT_R8, DRM_FORMAT_R8, DRM_FORMAT_R8 } },
+
+ // All these DRM_FORMATs were added in the same libdrm commit.
+#ifdef DRM_FORMAT_S010
+ { DRM_FORMAT_S010, VK_FORMAT_G16_B16_R16_3PLANE_420_UNORM, 3, { DRM_FORMAT_R16, DRM_FORMAT_R16, DRM_FORMAT_R16 } },
+ { DRM_FORMAT_S012, VK_FORMAT_G16_B16_R16_3PLANE_420_UNORM, 3, { DRM_FORMAT_R16, DRM_FORMAT_R16, DRM_FORMAT_R16 } },
+ { DRM_FORMAT_S016, VK_FORMAT_G16_B16_R16_3PLANE_420_UNORM, 3, { DRM_FORMAT_R16, DRM_FORMAT_R16, DRM_FORMAT_R16 } },
+ { DRM_FORMAT_S210, VK_FORMAT_G16_B16_R16_3PLANE_422_UNORM, 3, { DRM_FORMAT_R16, DRM_FORMAT_R16, DRM_FORMAT_R16 } },
+ { DRM_FORMAT_S212, VK_FORMAT_G16_B16_R16_3PLANE_422_UNORM, 3, { DRM_FORMAT_R16, DRM_FORMAT_R16, DRM_FORMAT_R16 } },
+ { DRM_FORMAT_S216, VK_FORMAT_G16_B16_R16_3PLANE_422_UNORM, 3, { DRM_FORMAT_R16, DRM_FORMAT_R16, DRM_FORMAT_R16 } },
+ { DRM_FORMAT_S410, VK_FORMAT_G16_B16_R16_3PLANE_444_UNORM, 3, { DRM_FORMAT_R16, DRM_FORMAT_R16, DRM_FORMAT_R16 } },
+ { DRM_FORMAT_S412, VK_FORMAT_G16_B16_R16_3PLANE_444_UNORM, 3, { DRM_FORMAT_R16, DRM_FORMAT_R16, DRM_FORMAT_R16 } },
+ { DRM_FORMAT_S416, VK_FORMAT_G16_B16_R16_3PLANE_444_UNORM, 3, { DRM_FORMAT_R16, DRM_FORMAT_R16, DRM_FORMAT_R16 } },
#endif
};
@@ -3318,6 +3414,51 @@ static inline VkFormat drm_to_vulkan_fmt
return VK_FORMAT_UNDEFINED;
}
+static inline int decompose_drm_frame_desc(AVDRMFrameDescriptor *dst,
+ const AVDRMFrameDescriptor *src)
+{
+ int map_idx = -1;
+ int p = 0;
+
+ if (!src || !dst)
+ return AVERROR(EINVAL);
+
+ for (int i = 0; i < FF_ARRAY_ELEMS(vulkan_drm_format_map); i++) {
+ if (vulkan_drm_format_map[i].drm_fourcc == src->layers[0].format) {
+ map_idx = i;
+ break;
+ }
+ }
+ if (map_idx == -1)
+ return AVERROR(ENOSYS);
+ if (vulkan_drm_format_map[map_idx].nb_planes == 1 &&
+ vulkan_drm_format_map[map_idx].fallbacks[0] == src->layers[0].format) {
+ return AVERROR(ENOSYS);
+ }
+
+ dst->nb_objects = src->nb_objects;
+ memcpy(dst->objects, src->objects, sizeof(src->objects));
+
+ for (int i = 0; i < src->nb_layers; i++) {
+ for (int j = 0; j < src->layers[i].nb_planes; j++) {
+ if (p >= vulkan_drm_format_map[map_idx].nb_planes)
+ goto end;
+
+ dst->layers[p].format = vulkan_drm_format_map[map_idx].fallbacks[p];
+ dst->layers[p].nb_planes = 1;
+ dst->layers[p].planes[0] = src->layers[i].planes[j];
+ p++;
+ }
+ }
+
+end:
+ dst->nb_layers = p;
+ return 0;
+}
+
+static VkImageAspectFlags plane_index_to_aspect(int plane);
+static VkImageAspectFlags plane_index_to_aspect_linear(int plane, int nb_images, int nb_planes);
+
static int vulkan_map_from_drm_frame_desc(AVHWFramesContext *hwfc, AVVkFrame **frame,
const AVFrame *src, int flags)
{
@@ -3329,10 +3470,15 @@ static int vulkan_map_from_drm_frame_des
VulkanDevicePriv *p = ctx->hwctx;
AVVulkanDeviceContext *hwctx = &p->p;
FFVulkanFunctions *vk = &p->vkctx.vkfn;
- const AVDRMFrameDescriptor *desc = (AVDRMFrameDescriptor *)src->data[0];
+ AVDRMFrameDescriptor desc_decomposed = { 0 };
+ AVDRMFrameDescriptor *desc_src = (AVDRMFrameDescriptor *)src->data[0];
+ AVDRMFrameDescriptor *desc = desc_src;
VkBindImageMemoryInfo bind_info[AV_DRM_MAX_PLANES];
VkBindImagePlaneMemoryInfo plane_info[AV_DRM_MAX_PLANES];
+ const int has_modifiers = !!(p->vkctx.extensions & FF_VK_EXT_DRM_MODIFIER_FLAGS);
+ int retry = 0;
+start:
for (int i = 0; i < desc->nb_layers; i++) {
if (drm_to_vulkan_fmt(desc->layers[i].format) == VK_FORMAT_UNDEFINED) {
av_log(ctx, AV_LOG_ERROR, "Unsupported DMABUF layer format %#08x!\n",
@@ -3341,13 +3487,21 @@ static int vulkan_map_from_drm_frame_des
}
}
+ if (!has_modifiers &&
+ desc->objects[0].format_modifier != DRM_FORMAT_MOD_LINEAR &&
+ desc->objects[0].format_modifier != DRM_FORMAT_MOD_INVALID) {
+ av_log(ctx, AV_LOG_ERROR, "Cannot map non-linear tiling image from DRM!\n");
+ err = AVERROR_EXTERNAL;
+ goto fail;
+ }
+
if (!(f = av_vk_frame_alloc())) {
av_log(ctx, AV_LOG_ERROR, "Unable to allocate memory for AVVkFrame!\n");
err = AVERROR(ENOMEM);
goto fail;
}
- f->tiling = VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT;
+ f->tiling = has_modifiers ? VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT : VK_IMAGE_TILING_LINEAR;
for (int i = 0; i < desc->nb_layers; i++) {
const int planes = desc->layers[i].nb_planes;
@@ -3385,7 +3539,7 @@ static int vulkan_map_from_drm_frame_des
.mipLevels = 1,
.arrayLayers = 1,
.flags = 0x0,
- .tiling = VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT,
+ .tiling = f->tiling,
.initialLayout = VK_IMAGE_LAYOUT_UNDEFINED, /* specs say so */
.usage = 0x0, /* filled in below */
.samples = VK_SAMPLE_COUNT_1_BIT,
@@ -3426,7 +3580,7 @@ static int vulkan_map_from_drm_frame_des
fmt_props = (VkPhysicalDeviceImageFormatInfo2) {
.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_IMAGE_FORMAT_INFO_2,
- .pNext = &props_ext,
+ .pNext = has_modifiers ? &props_ext : NULL,
.format = create_info.format,
.type = create_info.imageType,
.tiling = create_info.tiling,
@@ -3438,6 +3592,13 @@ static int vulkan_map_from_drm_frame_des
ret = vk->GetPhysicalDeviceImageFormatProperties2(hwctx->phys_dev,
&fmt_props, &props_ret);
if (ret != VK_SUCCESS) {
+ /* Decompose the AVDRMFrameDescriptor and retry */
+ if (!retry && !decompose_drm_frame_desc(&desc_decomposed, desc_src)) {
+ retry++;
+ desc = &desc_decomposed;
+ vulkan_frame_free(hwfc, f);
+ goto start;
+ }
av_log(ctx, AV_LOG_ERROR, "Cannot map DRM frame to Vulkan: %s\n",
ff_vk_ret2str(ret));
err = AVERROR_EXTERNAL;
@@ -3549,9 +3710,12 @@ static int vulkan_map_from_drm_frame_des
for (int i = 0; i < desc->nb_layers; i++) {
const int planes = desc->layers[i].nb_planes;
for (int j = 0; j < planes; j++) {
- VkImageAspectFlagBits aspect = j == 0 ? VK_IMAGE_ASPECT_MEMORY_PLANE_0_BIT_EXT :
- j == 1 ? VK_IMAGE_ASPECT_MEMORY_PLANE_1_BIT_EXT :
- VK_IMAGE_ASPECT_MEMORY_PLANE_2_BIT_EXT;
+ VkImageAspectFlagBits aspect;
+
+ if (has_modifiers)
+ aspect = plane_index_to_aspect(j);
+ else
+ aspect = plane_index_to_aspect_linear(j, desc->nb_layers, planes);
plane_info[bind_counts].sType = VK_STRUCTURE_TYPE_BIND_IMAGE_PLANE_MEMORY_INFO;
plane_info[bind_counts].pNext = NULL;
@@ -3592,15 +3756,20 @@ static int vulkan_map_from_drm_frame_syn
const AVDRMFrameDescriptor *desc, int flags)
{
int err;
+#ifdef DMA_BUF_IOCTL_EXPORT_SYNC_FILE
VkResult ret;
+#endif
AVHWDeviceContext *ctx = hwfc->device_ctx;
VulkanDevicePriv *p = ctx->hwctx;
VulkanFramesPriv *fp = hwfc->hwctx;
AVVulkanDeviceContext *hwctx = &p->p;
+#ifdef DMA_BUF_IOCTL_EXPORT_SYNC_FILE
FFVulkanFunctions *vk = &p->vkctx.vkfn;
+#endif
#ifdef DMA_BUF_IOCTL_EXPORT_SYNC_FILE
- if (p->vkctx.extensions & FF_VK_EXT_EXTERNAL_FD_SEM) {
+ if ((p->vkctx.extensions & FF_VK_EXT_EXTERNAL_FD_SEM) &&
+ !p->drm_export_sync_file_ret) {
VkCommandBuffer cmd_buf;
FFVkExecContext *exec;
VkImageMemoryBarrier2 img_bar[AV_NUM_DATA_POINTERS];
@@ -3618,18 +3787,19 @@ static int vulkan_map_from_drm_frame_syn
};
VkImportSemaphoreFdInfoKHR import_info;
struct dma_buf_export_sync_file implicit_fd_info = {
- .flags = DMA_BUF_SYNC_READ,
+ .flags = DMA_BUF_SYNC_RW,
.fd = -1,
};
- if (ioctl(desc->objects[i].fd, DMA_BUF_IOCTL_EXPORT_SYNC_FILE,
- &implicit_fd_info)) {
+ if (p->drm_export_sync_file_ret = ioctl(desc->objects[i].fd,
+ DMA_BUF_IOCTL_EXPORT_SYNC_FILE,
+ &implicit_fd_info)) {
err = AVERROR(errno);
- av_log(hwctx, AV_LOG_ERROR, "Failed to retrieve implicit DRM sync file: %s\n",
- av_err2str(err));
+ av_log(hwctx, AV_LOG_WARNING, "Failed to retrieve implicit DRM sync file: %s, "
+ "disabling for future frames\n", av_err2str(err));
for (; i >= 0; i--)
vk->DestroySemaphore(hwctx->act_dev, drm_sync_sem[i], hwctx->alloc);
- return err;
+ goto fallback;
}
ret = vk->CreateSemaphore(hwctx->act_dev, &sem_spawn,
@@ -3638,6 +3808,7 @@ static int vulkan_map_from_drm_frame_syn
av_log(hwctx, AV_LOG_ERROR, "Failed to create semaphore: %s\n",
ff_vk_ret2str(ret));
err = AVERROR_EXTERNAL;
+ close(implicit_fd_info.fd);
for (; i >= 0; i--)
vk->DestroySemaphore(hwctx->act_dev, drm_sync_sem[i], hwctx->alloc);
return err;
@@ -3656,6 +3827,7 @@ static int vulkan_map_from_drm_frame_syn
av_log(hwctx, AV_LOG_ERROR, "Failed to import semaphore: %s\n",
ff_vk_ret2str(ret));
err = AVERROR_EXTERNAL;
+ close(implicit_fd_info.fd);
for (; i >= 0; i--)
vk->DestroySemaphore(hwctx->act_dev, drm_sync_sem[i], hwctx->alloc);
return err;
@@ -3672,13 +3844,13 @@ static int vulkan_map_from_drm_frame_syn
drm_sync_sem, desc->nb_objects,
VK_PIPELINE_STAGE_2_ALL_COMMANDS_BIT, 1);
if (err < 0)
- return err;
+ goto fail;
err = ff_vk_exec_add_dep_frame(&p->vkctx, exec, dst,
VK_PIPELINE_STAGE_2_NONE,
VK_PIPELINE_STAGE_2_ALL_COMMANDS_BIT);
if (err < 0)
- return err;
+ goto fail;
ff_vk_frame_barrier(&p->vkctx, exec, dst, img_bar, &nb_img_bar,
VK_PIPELINE_STAGE_2_NONE,
@@ -3698,14 +3870,22 @@ static int vulkan_map_from_drm_frame_syn
err = ff_vk_exec_submit(&p->vkctx, exec);
if (err < 0)
- return err;
+ goto fail;
+
+ return 0;
+fail:
+ ff_vk_exec_discard_deps(&p->vkctx, exec);
+ return err;
} else
#endif
{
- AVVkFrame *f = (AVVkFrame *)dst->data[0];
- av_log(hwctx, AV_LOG_WARNING, "No support for synchronization when importing DMA-BUFs, "
- "image may be corrupted.\n");
- err = prepare_frame(hwfc, &fp->compute_exec, f, PREP_MODE_EXTERNAL_IMPORT);
+#ifdef DMA_BUF_IOCTL_EXPORT_SYNC_FILE
+fallback:
+#endif
+ av_log(hwctx, AV_LOG_DEBUG, "No support for synchronization when importing DMA-BUFs, "
+ "image may be corrupted.\n");
+ err = prepare_frame(hwfc, &fp->compute_exec,
+ (AVVkFrame *)dst->data[0], PREP_MODE_EXTERNAL_IMPORT);
if (err)
return err;
}
@@ -3742,7 +3922,7 @@ static int vulkan_map_from_drm(AVHWFrame
return 0;
fail:
- vulkan_frame_free(hwfc->device_ctx->hwctx, f);
+ vulkan_frame_free(hwfc, f);
dst->data[0] = NULL;
return err;
}
@@ -4112,16 +4292,10 @@ static int vulkan_map_to(AVHWFramesConte
#if CONFIG_LIBDRM
#if CONFIG_VAAPI
case AV_PIX_FMT_VAAPI:
- if (p->vkctx.extensions & FF_VK_EXT_DRM_MODIFIER_FLAGS)
- return vulkan_map_from_vaapi(hwfc, dst, src, flags);
- else
- return AVERROR(ENOSYS);
+ return vulkan_map_from_vaapi(hwfc, dst, src, flags);
#endif
case AV_PIX_FMT_DRM_PRIME:
- if (p->vkctx.extensions & FF_VK_EXT_DRM_MODIFIER_FLAGS)
- return vulkan_map_from_drm(hwfc, dst, src, flags);
- else
- return AVERROR(ENOSYS);
+ return vulkan_map_from_drm(hwfc, dst, src, flags);
#endif
default:
return AVERROR(ENOSYS);
@@ -4137,10 +4311,51 @@ typedef struct VulkanDRMMapping {
static void vulkan_unmap_to_drm(AVHWFramesContext *hwfc, HWMapDescriptor *hwmap)
{
AVDRMFrameDescriptor *drm_desc = hwmap->priv;
+#ifdef DMA_BUF_IOCTL_EXPORT_SYNC_FILE
+ AVVkFrame *f = (AVVkFrame *)hwmap->source->data[0];
+ AVHWDeviceContext *ctx = hwfc->device_ctx;
+ VulkanDevicePriv *p = ctx->hwctx;
+ AVVulkanDeviceContext *hwctx = &p->p;
+ FFVulkanFunctions *vk = &p->vkctx.vkfn;
+#endif
+
+ /* On unmap from DRM, import any fences the external consumer attached
+ * back into the original Vulkan frame's semaphore, so Vulkan waits on
+ * them before reusing the frame. We don't know if the consumer did reads
+ * or writes, so export both. */
+#ifdef DMA_BUF_IOCTL_EXPORT_SYNC_FILE
+ if ((p->vkctx.extensions & FF_VK_EXT_EXTERNAL_FD_SEM) &&
+ !p->drm_export_sync_file_ret) {
+ for (int i = 0; i < drm_desc->nb_objects; i++) {
+ struct dma_buf_export_sync_file implicit_fd_info = {
+ .flags = DMA_BUF_SYNC_RW,
+ .fd = -1,
+ };
- /* on unmap from DRM, make sure to import sync objects so that we are sync'd with any work that was
- * done on the buffer while exported. We don't know if who used the dmabuf did reads or writes, so protect against both */
- vulkan_map_from_drm_frame_sync(hwfc, hwmap->source, drm_desc, AV_HWFRAME_MAP_READ | AV_HWFRAME_MAP_WRITE);
+ if (drm_desc->objects[i].fd < 0)
+ break;
+
+ if (p->drm_export_sync_file_ret = ioctl(drm_desc->objects[i].fd,
+ DMA_BUF_IOCTL_EXPORT_SYNC_FILE,
+ &implicit_fd_info)) {
+ av_log(hwfc, AV_LOG_WARNING, "DMA_BUF_IOCTL_EXPORT_SYNC_FILE failed: %s, "
+ "disabling for future frames\n", av_err2str(AVERROR(errno)));
+ break;
+ }
+
+ VkImportSemaphoreFdInfoKHR import_info = {
+ .sType = VK_STRUCTURE_TYPE_IMPORT_SEMAPHORE_FD_INFO_KHR,
+ .semaphore = f->sem[i],
+ .handleType = VK_EXTERNAL_SEMAPHORE_HANDLE_TYPE_SYNC_FD_BIT,
+ .flags = VK_SEMAPHORE_IMPORT_TEMPORARY_BIT,
+ .fd = implicit_fd_info.fd,
+ };
+
+ if (vk->ImportSemaphoreFdKHR(hwctx->act_dev, &import_info) != VK_SUCCESS)
+ close(implicit_fd_info.fd);
+ }
+ }
+#endif
for (int i = 0; i < drm_desc->nb_objects; i++)
close(drm_desc->objects[i].fd);
@@ -4167,6 +4382,20 @@ static VkImageAspectFlags plane_index_to
return VK_IMAGE_ASPECT_MEMORY_PLANE_0_BIT_EXT;
}
+static VkImageAspectFlags plane_index_to_aspect_linear(int plane, int nb_images, int nb_planes)
+{
+ if (nb_images == 1 && nb_planes > 1) {
+ if (plane == 0) return VK_IMAGE_ASPECT_PLANE_0_BIT;
+ if (plane == 1) return VK_IMAGE_ASPECT_PLANE_1_BIT;
+ if (plane == 2) return VK_IMAGE_ASPECT_PLANE_2_BIT;
+
+ av_assert2(0 && "Invalid plane index");
+ return VK_IMAGE_ASPECT_PLANE_0_BIT;
+ }
+
+ return VK_IMAGE_ASPECT_COLOR_BIT;
+}
+
#ifdef DMA_BUF_IOCTL_EXPORT_SYNC_FILE
static int vulkan_drm_export_sync_fd(AVHWFramesContext *hwfc, AVVkFrame *f,
VulkanFramesPriv *fp, int nb_sems)
@@ -4245,14 +4474,22 @@ static int vulkan_map_to_drm(AVHWFramesC
VulkanFramesPriv *fp = hwfc->hwctx;
const int planes = av_pix_fmt_count_planes(hwfc->sw_format);
const int nb_images = ff_vk_count_images(f);
+ const int is_modifier_tiling = (p->vkctx.extensions & FF_VK_EXT_DRM_MODIFIER_FLAGS) &&
+ f->tiling == VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT;
VkImageDrmFormatModifierPropertiesEXT drm_mod = {
.sType = VK_STRUCTURE_TYPE_IMAGE_DRM_FORMAT_MODIFIER_PROPERTIES_EXT,
};
const int nb_sems = nb_images;
int free_drm_desc_on_err = 1;
int sync_fd = -1;
+ AVDRMFrameDescriptor *drm_desc = NULL;
- AVDRMFrameDescriptor *drm_desc = av_mallocz(sizeof(*drm_desc));
+ if (f->tiling == VK_IMAGE_TILING_OPTIMAL) {
+ av_log(hwfc, AV_LOG_ERROR, "Cannot map optimal tiling image to DRM!\n");
+ return AVERROR(EINVAL);
+ }
+
+ drm_desc = av_mallocz(sizeof(*drm_desc));
if (!drm_desc)
return AVERROR(ENOMEM);
@@ -4261,8 +4498,9 @@ static int vulkan_map_to_drm(AVHWFramesC
goto end;
#ifdef DMA_BUF_IOCTL_EXPORT_SYNC_FILE
- if ((p->vkctx.extensions & FF_VK_EXT_EXTERNAL_FD_SEM) &&
- f->tiling == VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT &&
+ // Known to be not working on AMD GPU with VAAPI, disable it for now
+ if (0 && (p->vkctx.extensions & FF_VK_EXT_EXTERNAL_FD_SEM) &&
+ !p->drm_import_sync_file_ret &&
vk->GetSemaphoreFdKHR && vk->CreateSemaphore) {
err = vulkan_drm_export_sync_fd(hwfc, f, fp, nb_sems);
if (err < 0)
@@ -4290,13 +4528,16 @@ static int vulkan_map_to_drm(AVHWFramesC
/* It will be freed in ff_hwframe_map_create callback */
free_drm_desc_on_err = 0;
- ret = vk->GetImageDrmFormatModifierPropertiesEXT(hwctx->act_dev, f->img[0],
- &drm_mod);
- if (ret != VK_SUCCESS) {
- av_log(hwfc, AV_LOG_ERROR, "Failed to retrieve DRM format modifier!\n");
- err = AVERROR_EXTERNAL;
- goto end;
- }
+ if (is_modifier_tiling) {
+ ret = vk->GetImageDrmFormatModifierPropertiesEXT(hwctx->act_dev, f->img[0],
+ &drm_mod);
+ if (ret != VK_SUCCESS) {
+ av_log(hwfc, AV_LOG_ERROR, "Failed to retrieve DRM format modifier!\n");
+ err = AVERROR_EXTERNAL;
+ goto end;
+ }
+ } else
+ drm_mod.drmFormatModifier = DRM_FORMAT_MOD_LINEAR;
for (int i = 0; (i < planes) && (f->mem[i]); i++) {
VkMemoryGetFdInfoKHR export_info = {
@@ -4313,16 +4554,20 @@ static int vulkan_map_to_drm(AVHWFramesC
goto end;
}
-#if HAVE_LINUX_DMA_BUF_H && defined(DMA_BUF_IOCTL_IMPORT_SYNC_FILE)
- if (sync_fd >= 0) {
+#ifdef DMA_BUF_IOCTL_IMPORT_SYNC_FILE
+ // Known to be not working on AMD GPU with VAAPI, disable it for now
+ if (0 && sync_fd >= 0 && !p->drm_import_sync_file_ret) {
int dup_fd = dup(sync_fd);
if (dup_fd >= 0) {
struct dma_buf_import_sync_file import_info = {
- .flags = DMA_BUF_SYNC_WRITE,
+ .flags = DMA_BUF_SYNC_RW,
.fd = dup_fd,
};
- if (ioctl(drm_desc->objects[i].fd, DMA_BUF_IOCTL_IMPORT_SYNC_FILE, &import_info) < 0)
- av_log(hwfc, AV_LOG_WARNING, "DMA_BUF_IOCTL_IMPORT_SYNC_FILE failed: %s\n", av_err2str(AVERROR(errno)));
+ if (p->drm_import_sync_file_ret = ioctl(drm_desc->objects[i].fd,
+ DMA_BUF_IOCTL_IMPORT_SYNC_FILE,
+ &import_info))
+ av_log(hwfc, AV_LOG_WARNING, "DMA_BUF_IOCTL_IMPORT_SYNC_FILE failed: %s, "
+ "disabling for future frames\n", av_err2str(AVERROR(errno)));
close(dup_fd);
} else {
av_log(hwfc, AV_LOG_WARNING, "dup(sync_fd) failed: %s\n", av_err2str(AVERROR(errno)));
@@ -4341,8 +4586,13 @@ static int vulkan_map_to_drm(AVHWFramesC
VkFormat plane_vkfmt = av_vkfmt_from_pixfmt(hwfc->sw_format)[i];
drm_desc->layers[i].format = vulkan_fmt_to_drm(plane_vkfmt);
- drm_desc->layers[i].nb_planes = fp->drm_format_modifier_properties[i].drmFormatModifierPlaneCount;
+ drm_desc->layers[i].nb_planes = is_modifier_tiling ? fp->drm_format_modifier_properties[i].drmFormatModifierPlaneCount : 1;
+ if (drm_desc->layers[i].format == DRM_FORMAT_INVALID) {
+ av_log(hwfc, AV_LOG_ERROR, "Cannot map to DRM layer, unsupported!\n");
+ err = AVERROR_PATCHWELCOME;
+ goto end;
+ }
if (drm_desc->layers[i].nb_planes > MAX_MEMORY_PLANES) {
av_log(hwfc, AV_LOG_ERROR, "Too many memory planes for DRM format!\n");
err = AVERROR_EXTERNAL;
@@ -4351,10 +4601,13 @@ static int vulkan_map_to_drm(AVHWFramesC
for (int j = 0; j < drm_desc->layers[i].nb_planes; j++) {
VkSubresourceLayout layout;
- int aspect_plane = (nb_images == 1) ? i : j;
- VkImageSubresource sub = {
- .aspectMask = plane_index_to_aspect(aspect_plane),
- };
+ VkImageSubresource sub = { 0 };
+
+ if (is_modifier_tiling) {
+ int aspect_plane = (nb_images == 1) ? i : j;
+ sub.aspectMask = plane_index_to_aspect(aspect_plane);
+ } else
+ sub.aspectMask = plane_index_to_aspect_linear(i, nb_images, planes);
drm_desc->layers[i].planes[j].object_index = FFMIN(i, drm_desc->nb_objects - 1);
@@ -4362,34 +4615,24 @@ static int vulkan_map_to_drm(AVHWFramesC
drm_desc->layers[i].planes[j].offset = layout.offset;
drm_desc->layers[i].planes[j].pitch = layout.rowPitch;
}
-
- if (drm_desc->layers[i].format == DRM_FORMAT_INVALID) {
- av_log(hwfc, AV_LOG_ERROR, "Cannot map to DRM layer, unsupported!\n");
- err = AVERROR_PATCHWELCOME;
- goto end;
- }
-
-
- if (f->tiling == VK_IMAGE_TILING_OPTIMAL)
- continue;
-
}
dst->width = src->width;
dst->height = src->height;
dst->data[0] = (uint8_t *)drm_desc;
- dst->hw_frames_ctx = av_buffer_ref(src->hw_frames_ctx);
if (sync_fd >= 0)
close(sync_fd);
- av_log(hwfc, AV_LOG_VERBOSE, "Mapped AVVkFrame to a DRM object!\n");
+ av_log(hwfc, AV_LOG_DEBUG, "Mapped AVVkFrame to a DRM object!\n");
return 0;
end:
- for (int i = 0; i < drm_desc->nb_objects; i++)
+ for (int i = 0; i < drm_desc->nb_objects; i++) {
close(drm_desc->objects[i].fd);
+ drm_desc->objects[i].fd = -1;
+ }
if (free_drm_desc_on_err)
av_free(drm_desc);
if (sync_fd >= 0)
@@ -4433,16 +4676,10 @@ static int vulkan_map_from(AVHWFramesCon
switch (dst->format) {
#if CONFIG_LIBDRM
case AV_PIX_FMT_DRM_PRIME:
- if (p->vkctx.extensions & FF_VK_EXT_DRM_MODIFIER_FLAGS)
- return vulkan_map_to_drm(hwfc, dst, src, flags);
- else
- return AVERROR(ENOSYS);
+ return vulkan_map_to_drm(hwfc, dst, src, flags);
#if CONFIG_VAAPI
case AV_PIX_FMT_VAAPI:
- if (p->vkctx.extensions & FF_VK_EXT_DRM_MODIFIER_FLAGS)
- return vulkan_map_to_vaapi(hwfc, dst, src, flags);
- else
- return AVERROR(ENOSYS);
+ return vulkan_map_to_vaapi(hwfc, dst, src, flags);
#endif
#endif
default:
@@ -2,7 +2,7 @@ Index: FFmpeg/libavutil/hwcontext_vulkan.c
=================================================================== ===================================================================
--- FFmpeg.orig/libavutil/hwcontext_vulkan.c --- FFmpeg.orig/libavutil/hwcontext_vulkan.c
+++ FFmpeg/libavutil/hwcontext_vulkan.c +++ FFmpeg/libavutil/hwcontext_vulkan.c
@@ -968,7 +968,7 @@ static const char *vk_dev_type(enum VkPh @@ -1355,7 +1355,7 @@ static const char *vk_dev_type(enum VkPh
static int find_device(AVHWDeviceContext *ctx, VulkanDeviceSelection *select) static int find_device(AVHWDeviceContext *ctx, VulkanDeviceSelection *select)
{ {
int err = 0, choice = -1; int err = 0, choice = -1;
@@ -11,7 +11,7 @@ Index: FFmpeg/libavutil/hwcontext_vulkan.c
VkResult ret; VkResult ret;
VulkanDevicePriv *p = ctx->hwctx; VulkanDevicePriv *p = ctx->hwctx;
AVVulkanDeviceContext *hwctx = &p->p; AVVulkanDeviceContext *hwctx = &p->p;
@@ -1035,63 +1035,78 @@ static int find_device(AVHWDeviceContext @@ -1431,63 +1431,78 @@ static int find_device(AVHWDeviceContext
if (select->has_uuid) { if (select->has_uuid) {
for (int i = 0; i < num; i++) { for (int i = 0; i < num; i++) {
@@ -0,0 +1,94 @@
Index: FFmpeg/libavfilter/vf_hwmap.c
===================================================================
--- FFmpeg.orig/libavfilter/vf_hwmap.c
+++ FFmpeg/libavfilter/vf_hwmap.c
@@ -26,6 +26,12 @@
#include "libavutil/hwcontext_d3d11va.h"
#endif
+#if CONFIG_VULKAN
+#include "libavutil/hwcontext_vulkan.h"
+#include "libavutil/vulkan_loader.h"
+#include "libavutil/vulkan.h"
+#endif
+
#include "avfilter.h"
#include "filters.h"
#include "formats.h"
@@ -39,6 +45,11 @@ typedef struct HWMapContext {
int mode;
char *derive_device_type;
int reverse;
+
+#if CONFIG_VULKAN
+ FFVulkanFunctions vkfn;
+ int vkfn_loaded;
+#endif
} HWMapContext;
static int hwmap_query_formats(const AVFilterContext *avctx,
@@ -193,6 +204,18 @@ static int hwmap_config_output(AVFilterL
goto fail;
}
+#if CONFIG_VULKAN
+ if (inl->hw_frames_ctx && inlink->format == AV_PIX_FMT_VULKAN) {
+ AVHWFramesContext *hwfc_src = (AVHWFramesContext *)inl->hw_frames_ctx->data;
+ AVVulkanFramesContext *vkfc_src = hwfc_src->hwctx;
+ AVHWFramesContext *hwfc_dst = (AVHWFramesContext *)source->data;
+ AVVulkanFramesContext *vkfc_dst = hwfc_dst->hwctx;
+
+ // Passthrough the VK_IMAGE_USAGE_*_BIT
+ vkfc_dst->usage = vkfc_src->usage;
+ }
+#endif
+
// Here is the naughty bit. This overwriting changes what
// ff_get_video_buffer() in the previous filter returns -
// it will now give a frame allocated here mapped back to
@@ -375,6 +398,45 @@ static int hwmap_filter_frame(AVFilterLi
if (err < 0)
goto fail;
+#if CONFIG_VULKAN
+ if (ctx->reverse &&
+ input->hw_frames_ctx && map->hw_frames_ctx &&
+ input->format == AV_PIX_FMT_VULKAN &&
+ (map->format == AV_PIX_FMT_VAAPI ||
+ map->format == AV_PIX_FMT_DRM_PRIME)) {
+ // If we mapped backwards from vulkan to drm_prime, we need
+ // to wait for the AVVkFrame semaphores to be signaled.
+ AVHWFramesContext *hwfc = (AVHWFramesContext *)input->hw_frames_ctx->data;
+ AVVulkanDeviceContext *vk_dev = hwfc->device_ctx->hwctx;
+ AVVkFrame *vkf = (AVVkFrame *)input->data[0];
+ const int nb_sems = vkf ? ff_vk_count_images(vkf) : 0;
+
+ if (hwfc->device_ctx->type != AV_HWDEVICE_TYPE_VULKAN || !vkf || !nb_sems)
+ goto exit;
+
+ if (!ctx->vkfn_loaded) {
+ uint64_t exts = ff_vk_extensions_to_mask(vk_dev->enabled_dev_extensions,
+ vk_dev->nb_enabled_dev_extensions);
+ err = ff_vk_load_functions(hwfc->device_ctx, &ctx->vkfn, exts, 1, 1);
+ if (err < 0)
+ goto fail;
+ ctx->vkfn_loaded = 1;
+ }
+ if (ctx->vkfn.WaitSemaphores) {
+ VkSemaphoreWaitInfo wait_info = {
+ .sType = VK_STRUCTURE_TYPE_SEMAPHORE_WAIT_INFO,
+ .flags = 0x0,
+ .semaphoreCount = nb_sems,
+ .pSemaphores = vkf->sem,
+ .pValues = vkf->sem_value,
+ };
+ ctx->vkfn.WaitSemaphores(vk_dev->act_dev, &wait_info, UINT64_MAX);
+ av_log(avctx, AV_LOG_DEBUG, "Vulkan sems for reverse-mapped DRM objects signaled!\n");
+ }
+ }
+exit:
+#endif
+
av_frame_free(&input);
av_log(ctx, AV_LOG_DEBUG, "Filter output: %s, %ux%u (%"PRId64").\n",
@@ -0,0 +1,506 @@
Index: FFmpeg/libavfilter/vf_flip_vulkan.c
===================================================================
--- FFmpeg.orig/libavfilter/vf_flip_vulkan.c
+++ FFmpeg/libavfilter/vf_flip_vulkan.c
@@ -109,13 +109,13 @@ static av_cold int init_filter(AVFilterC
switch (type)
{
case FLIP_HORIZONTAL:
- GLSLF(2, vec4 res = imageLoad(input_image[%i], ivec2(size.x - pos.x, pos.y)); ,i);
+ GLSLF(2, vec4 res = imageLoad(input_image[%i], ivec2(size.x - 1 - pos.x, pos.y)); ,i);
break;
case FLIP_VERTICAL:
- GLSLF(2, vec4 res = imageLoad(input_image[%i], ivec2(pos.x, size.y - pos.y)); ,i);
+ GLSLF(2, vec4 res = imageLoad(input_image[%i], ivec2(pos.x, size.y - 1 - pos.y)); ,i);
break;
case FLIP_BOTH:
- GLSLF(2, vec4 res = imageLoad(input_image[%i], ivec2(size.xy - pos.xy));, i);
+ GLSLF(2, vec4 res = imageLoad(input_image[%i], ivec2(size.xy - 1 - pos.xy));, i);
break;
default:
GLSLF(2, vec4 res = imageLoad(input_image[%i], pos); ,i);
Index: FFmpeg/libavfilter/vf_overlay_vulkan.c
===================================================================
--- FFmpeg.orig/libavfilter/vf_overlay_vulkan.c
+++ FFmpeg/libavfilter/vf_overlay_vulkan.c
@@ -35,6 +35,7 @@ typedef struct OverlayVulkanContext {
FFVkExecPool e;
AVVulkanDeviceQueueFamily *qf;
FFVulkanShader shd;
+ FFVulkanShader shd_pass;
/* Push constants / options */
struct {
@@ -46,6 +47,10 @@ typedef struct OverlayVulkanContext {
int overlay_y;
int overlay_w;
int overlay_h;
+
+ int opt_repeatlast;
+ int opt_shortest;
+ int opt_eof_action;
} OverlayVulkanContext;
static const char overlay_noalpha[] = {
@@ -85,12 +90,13 @@ static av_cold int init_filter(AVFilterC
uint8_t *spv_data;
size_t spv_len;
void *spv_opaque = NULL;
+ void *spv_opaque_pass = NULL;
OverlayVulkanContext *s = ctx->priv;
FFVulkanContext *vkctx = &s->vkctx;
const int planes = av_pix_fmt_count_planes(s->vkctx.output_format);
const int ialpha = av_pix_fmt_desc_get(s->vkctx.input_format)->flags & AV_PIX_FMT_FLAG_ALPHA;
const AVPixFmtDescriptor *pix_desc = av_pix_fmt_desc_get(s->vkctx.output_format);
- FFVulkanShader *shd = &s->shd;
+ FFVulkanShader *shd;
FFVkSPIRVCompiler *spv;
FFVulkanDescriptorSetBinding *desc;
@@ -108,12 +114,15 @@ static av_cold int init_filter(AVFilterC
}
RET(ff_vk_exec_pool_init(vkctx, s->qf, &s->e, s->qf->num*4, 0, 0, 0, NULL));
+
+ /* overlay */
RET(ff_vk_shader_init(vkctx, &s->shd, "overlay",
VK_SHADER_STAGE_COMPUTE_BIT,
NULL, 0,
32, 32, 1,
0));
+ shd = &s->shd;
GLSLC(0, layout(push_constant, std430) uniform pushConstants { );
GLSLC(1, ivec2 o_offset[3]; );
GLSLC(1, ivec2 o_size[3]; );
@@ -169,9 +178,9 @@ static av_cold int init_filter(AVFilterC
GLSLC(1, } );
GLSLC(0, } );
- RET(spv->compile_shader(vkctx, spv, shd, &spv_data, &spv_len, "main",
+ RET(spv->compile_shader(vkctx, spv, &s->shd, &spv_data, &spv_len, "main",
&spv_opaque));
- RET(ff_vk_shader_link(vkctx, shd, spv_data, spv_len, "main"));
+ RET(ff_vk_shader_link(vkctx, &s->shd, spv_data, spv_len, "main"));
RET(ff_vk_shader_register_exec(vkctx, &s->e, &s->shd));
@@ -189,11 +198,59 @@ static av_cold int init_filter(AVFilterC
s->opts.o_size[4] = s->opts.o_size[0] >> pix_desc->log2_chroma_w;
s->opts.o_size[5] = s->opts.o_size[1] >> pix_desc->log2_chroma_h;
+ /* overlay_pass */
+ RET(ff_vk_shader_init(vkctx, &s->shd_pass, "overlay_pass",
+ VK_SHADER_STAGE_COMPUTE_BIT,
+ NULL, 0,
+ 32, 32, 1,
+ 0));
+
+ desc = (FFVulkanDescriptorSetBinding []) {
+ {
+ .name = "main_img",
+ .type = VK_DESCRIPTOR_TYPE_STORAGE_IMAGE,
+ .mem_layout = ff_vk_shader_rep_fmt(s->vkctx.input_format, FF_VK_REP_FLOAT),
+ .mem_quali = "readonly",
+ .dimensions = 2,
+ .elems = planes,
+ .stages = VK_SHADER_STAGE_COMPUTE_BIT,
+ },
+ {
+ .name = "output_img",
+ .type = VK_DESCRIPTOR_TYPE_STORAGE_IMAGE,
+ .mem_layout = ff_vk_shader_rep_fmt(s->vkctx.output_format, FF_VK_REP_FLOAT),
+ .mem_quali = "writeonly",
+ .dimensions = 2,
+ .elems = planes,
+ .stages = VK_SHADER_STAGE_COMPUTE_BIT,
+ },
+ };
+
+ RET(ff_vk_shader_add_descriptor_set(vkctx, &s->shd_pass, desc, 2, 0, 0));
+
+ shd = &s->shd_pass;
+ GLSLC(0, void main() );
+ GLSLC(0, { );
+ GLSLC(1, ivec2 pos = ivec2(gl_GlobalInvocationID.xy); );
+ GLSLF(1, int planes = %i; ,planes);
+ GLSLC(1, for (int i = 0; i < planes; i++) { );
+ GLSLC(2, vec4 res = imageLoad(main_img[i], pos); );
+ GLSLC(2, imageStore(output_img[i], pos, res); );
+ GLSLC(1, } );
+ GLSLC(0, } );
+
+ RET(spv->compile_shader(vkctx, spv, &s->shd_pass, &spv_data, &spv_len, "main",
+ &spv_opaque_pass));
+ RET(ff_vk_shader_link(vkctx, &s->shd_pass, spv_data, spv_len, "main"));
+ RET(ff_vk_shader_register_exec(vkctx, &s->e, &s->shd_pass));
+
s->initialized = 1;
fail:
if (spv_opaque)
spv->free_shader(spv, &spv_opaque);
+ if (spv_opaque_pass)
+ spv->free_shader(spv, &spv_opaque_pass);
if (spv)
spv->uninit(&spv);
@@ -215,22 +272,11 @@ static int overlay_vulkan_blend(FFFrameS
if (err < 0)
goto fail;
- if (!input_main || !input_overlay)
- return 0;
-
- if (!s->initialized) {
- AVHWFramesContext *main_fc = (AVHWFramesContext*)input_main->hw_frames_ctx->data;
- AVHWFramesContext *overlay_fc = (AVHWFramesContext*)input_overlay->hw_frames_ctx->data;
- if (main_fc->sw_format != overlay_fc->sw_format) {
- av_log(ctx, AV_LOG_ERROR, "Mismatching sw formats!\n");
- return AVERROR(EINVAL);
- }
-
- s->overlay_w = input_overlay->width;
- s->overlay_h = input_overlay->height;
+ if (!input_main)
+ return AVERROR_BUG;
+ if (!s->initialized)
RET(init_filter(ctx));
- }
out = ff_get_video_buffer(outlink, outlink->w, outlink->h);
if (!out) {
@@ -238,9 +284,14 @@ static int overlay_vulkan_blend(FFFrameS
goto fail;
}
- RET(ff_vk_filter_process_Nin(&s->vkctx, &s->e, &s->shd,
- out, (AVFrame *[]){ input_main, input_overlay }, 2,
- VK_NULL_HANDLE, &s->opts, sizeof(s->opts)));
+ if (input_overlay)
+ RET(ff_vk_filter_process_Nin(&s->vkctx, &s->e, &s->shd,
+ out, (AVFrame *[]){ input_main, input_overlay }, 2,
+ VK_NULL_HANDLE, &s->opts, sizeof(s->opts)));
+ else /* passthrough */
+ RET(ff_vk_filter_process_simple(&s->vkctx, &s->e, &s->shd_pass,
+ out, input_main, VK_NULL_HANDLE,
+ &s->opts, sizeof(s->opts)));
err = av_frame_copy_props(out, input_main);
if (err < 0)
@@ -258,6 +309,20 @@ static int overlay_vulkan_config_output(
int err;
AVFilterContext *avctx = outlink->src;
OverlayVulkanContext *s = avctx->priv;
+ AVFilterLink *inlink = avctx->inputs[0];
+ AVFilterLink *inlink_overlay = avctx->inputs[1];
+ FilterLink *inl = ff_filter_link(inlink);
+ FilterLink *inl_overlay = ff_filter_link(inlink_overlay);
+ AVHWFramesContext *main_fc = (AVHWFramesContext*)inl->hw_frames_ctx->data;
+ AVHWFramesContext *overlay_fc = (AVHWFramesContext*)inl_overlay->hw_frames_ctx->data;
+
+ if (main_fc->sw_format != overlay_fc->sw_format) {
+ av_log(avctx, AV_LOG_ERROR, "Mismatching sw formats!\n");
+ return AVERROR(EINVAL);
+ }
+
+ s->overlay_w = inlink_overlay->w;
+ s->overlay_h = inlink_overlay->h;
err = ff_vk_filter_config_output(outlink);
if (err < 0)
@@ -267,6 +332,11 @@ static int overlay_vulkan_config_output(
if (err < 0)
return err;
+ s->fs.opt_repeatlast = s->opt_repeatlast;
+ s->fs.opt_shortest = s->opt_shortest;
+ s->fs.opt_eof_action = s->opt_eof_action;
+ s->fs.time_base = outlink->time_base = inlink->time_base;
+
return ff_framesync_configure(&s->fs);
}
@@ -293,6 +363,7 @@ static void overlay_vulkan_uninit(AVFilt
ff_vk_exec_pool_free(vkctx, &s->e);
ff_vk_shader_free(vkctx, &s->shd);
+ ff_vk_shader_free(vkctx, &s->shd_pass);
ff_vk_uninit(&s->vkctx);
ff_framesync_uninit(&s->fs);
@@ -305,6 +376,14 @@ static void overlay_vulkan_uninit(AVFilt
static const AVOption overlay_vulkan_options[] = {
{ "x", "Set horizontal offset", OFFSET(overlay_x), AV_OPT_TYPE_INT, {.i64 = 0}, 0, INT_MAX, .flags = FLAGS },
{ "y", "Set vertical offset", OFFSET(overlay_y), AV_OPT_TYPE_INT, {.i64 = 0}, 0, INT_MAX, .flags = FLAGS },
+ { "eof_action", "Action to take when encountering EOF from secondary input ",
+ OFFSET(opt_eof_action), AV_OPT_TYPE_INT, { .i64 = EOF_ACTION_REPEAT },
+ EOF_ACTION_REPEAT, EOF_ACTION_PASS, .flags = FLAGS, .unit = "eof_action" },
+ { "repeat", "Repeat the previous frame.", 0, AV_OPT_TYPE_CONST, { .i64 = EOF_ACTION_REPEAT }, .flags = FLAGS, .unit = "eof_action" },
+ { "endall", "End both streams.", 0, AV_OPT_TYPE_CONST, { .i64 = EOF_ACTION_ENDALL }, .flags = FLAGS, .unit = "eof_action" },
+ { "pass", "Pass through the main input.", 0, AV_OPT_TYPE_CONST, { .i64 = EOF_ACTION_PASS }, .flags = FLAGS, .unit = "eof_action" },
+ { "shortest", "force termination when the shortest input terminates", OFFSET(opt_shortest), AV_OPT_TYPE_BOOL, { .i64 = 0 }, 0, 1, FLAGS },
+ { "repeatlast", "repeat overlay of the last overlay frame", OFFSET(opt_repeatlast), AV_OPT_TYPE_BOOL, { .i64 = 1 }, 0, 1, FLAGS },
{ NULL },
};
Index: FFmpeg/libavfilter/vf_scale_vulkan.c
===================================================================
--- FFmpeg.orig/libavfilter/vf_scale_vulkan.c
+++ FFmpeg/libavfilter/vf_scale_vulkan.c
@@ -104,6 +104,23 @@ static const char write_nv12[] = {
C(0, } )
};
+static const char write_nv16[] = {
+ C(0, void write_nv16(vec4 src, ivec2 pos) )
+ C(0, { )
+ C(1, imageStore(output_img[0], pos, vec4(src.r, 0.0, 0.0, 0.0)); )
+ C(1, pos.x >>= 1; )
+ C(1, imageStore(output_img[1], pos, vec4(src.g, src.b, 0.0, 0.0)); )
+ C(0, } )
+};
+
+static const char write_nv24[] = {
+ C(0, void write_nv24(vec4 src, ivec2 pos) )
+ C(0, { )
+ C(1, imageStore(output_img[0], pos, vec4(src.r, 0.0, 0.0, 0.0)); )
+ C(1, imageStore(output_img[1], pos, vec4(src.g, src.b, 0.0, 0.0)); )
+ C(0, } )
+};
+
static const char write_420[] = {
C(0, void write_420(vec4 src, ivec2 pos) )
C(0, { )
@@ -114,6 +131,16 @@ static const char write_420[] = {
C(0, } )
};
+static const char write_422[] = {
+ C(0, void write_422(vec4 src, ivec2 pos) )
+ C(0, { )
+ C(1, imageStore(output_img[0], pos, vec4(src.r, 0.0, 0.0, 0.0)); )
+ C(1, pos.x >>= 1; )
+ C(1, imageStore(output_img[1], pos, vec4(src.g, 0.0, 0.0, 0.0)); )
+ C(1, imageStore(output_img[2], pos, vec4(src.b, 0.0, 0.0, 0.0)); )
+ C(0, } )
+};
+
static const char write_444[] = {
C(0, void write_444(vec4 src, ivec2 pos) )
C(0, { )
@@ -134,9 +161,30 @@ static int init_scale_shader(AVFilterCon
}
switch (s->vkctx.output_format) {
- case AV_PIX_FMT_NV12: GLSLD(write_nv12); break;
- case AV_PIX_FMT_YUV420P: GLSLD( write_420); break;
- case AV_PIX_FMT_YUV444P: GLSLD( write_444); break;
+ case AV_PIX_FMT_NV12:
+ case AV_PIX_FMT_P010:
+ case AV_PIX_FMT_P012:
+ case AV_PIX_FMT_P016: GLSLD(write_nv12); break;
+ case AV_PIX_FMT_NV16:
+ case AV_PIX_FMT_P210:
+ case AV_PIX_FMT_P212:
+ case AV_PIX_FMT_P216: GLSLD(write_nv16); break;
+ case AV_PIX_FMT_NV24:
+ case AV_PIX_FMT_P410:
+ case AV_PIX_FMT_P412:
+ case AV_PIX_FMT_P416: GLSLD(write_nv24); break;
+ case AV_PIX_FMT_YUV420P:
+ case AV_PIX_FMT_YUV420P10:
+ case AV_PIX_FMT_YUV420P12:
+ case AV_PIX_FMT_YUV420P16: GLSLD( write_420); break;
+ case AV_PIX_FMT_YUV422P:
+ case AV_PIX_FMT_YUV422P10:
+ case AV_PIX_FMT_YUV422P12:
+ case AV_PIX_FMT_YUV422P16: GLSLD( write_422); break;
+ case AV_PIX_FMT_YUV444P:
+ case AV_PIX_FMT_YUV444P10:
+ case AV_PIX_FMT_YUV444P12:
+ case AV_PIX_FMT_YUV444P16: GLSLD( write_444); break;
default: break;
}
@@ -149,7 +197,8 @@ static int init_scale_shader(AVFilterCon
GLSLC(1, vec2 c_o = vec2(crop_x, crop_y) / in_d; );
GLSLC(0, );
- if (s->vkctx.output_format == s->vkctx.input_format) {
+ if (s->vkctx.output_format == s->vkctx.input_format ||
+ !ff_vk_mt_is_np_rgb(s->vkctx.input_format)) {
for (int i = 0; i < desc[1].elems; i++) {
GLSLF(1, size = imageSize(output_img[%i]); ,i);
GLSLC(1, if (IS_WITHIN(pos, size)) { );
@@ -166,16 +215,38 @@ static int init_scale_shader(AVFilterCon
GLSLC(1, vec4 res = scale_bilinear(0, pos, c_r, c_o); );
GLSLF(1, res = rgb2yuv(res, %i); ,s->out_range == AVCOL_RANGE_JPEG);
switch (s->vkctx.output_format) {
- case AV_PIX_FMT_NV12: GLSLC(1, write_nv12(res, pos); ); break;
- case AV_PIX_FMT_YUV420P: GLSLC(1, write_420(res, pos); ); break;
- case AV_PIX_FMT_YUV444P: GLSLC(1, write_444(res, pos); ); break;
+ case AV_PIX_FMT_NV12:
+ case AV_PIX_FMT_P010:
+ case AV_PIX_FMT_P012:
+ case AV_PIX_FMT_P016: GLSLC(1, write_nv12(res, pos); ); break;
+ case AV_PIX_FMT_NV16:
+ case AV_PIX_FMT_P210:
+ case AV_PIX_FMT_P212:
+ case AV_PIX_FMT_P216: GLSLC(1, write_nv16(res, pos); ); break;
+ case AV_PIX_FMT_NV24:
+ case AV_PIX_FMT_P410:
+ case AV_PIX_FMT_P412:
+ case AV_PIX_FMT_P416: GLSLC(1, write_nv24(res, pos); ); break;
+ case AV_PIX_FMT_YUV420P:
+ case AV_PIX_FMT_YUV420P10:
+ case AV_PIX_FMT_YUV420P12:
+ case AV_PIX_FMT_YUV420P16: GLSLC(1, write_420(res, pos); ); break;
+ case AV_PIX_FMT_YUV422P:
+ case AV_PIX_FMT_YUV422P10:
+ case AV_PIX_FMT_YUV422P12:
+ case AV_PIX_FMT_YUV422P16: GLSLC(1, write_422(res, pos); ); break;
+ case AV_PIX_FMT_YUV444P:
+ case AV_PIX_FMT_YUV444P10:
+ case AV_PIX_FMT_YUV444P12:
+ case AV_PIX_FMT_YUV444P16: GLSLC(1, write_444(res, pos); ); break;
default: return AVERROR(EINVAL);
}
}
GLSLC(0, } );
- if (s->vkctx.output_format != s->vkctx.input_format) {
+ if (s->vkctx.output_format != s->vkctx.input_format &&
+ ff_vk_mt_is_np_rgb(s->vkctx.input_format)) {
const AVLumaCoefficients *lcoeffs;
double tmp_mat[3][3];
@@ -385,7 +456,8 @@ static int scale_vulkan_filter_frame(AVF
if (s->out_range != AVCOL_RANGE_UNSPECIFIED)
out->color_range = s->out_range;
- if (s->vkctx.output_format != s->vkctx.input_format)
+ if (s->vkctx.output_format != s->vkctx.input_format &&
+ ff_vk_mt_is_np_rgb(s->vkctx.input_format))
out->chroma_location = AVCHROMA_LOC_TOPLEFT;
av_frame_free(&in);
@@ -440,13 +512,58 @@ static int scale_vulkan_config_output(AV
return AVERROR_PATCHWELCOME;
}
} else if (s->vkctx.output_format != s->vkctx.input_format) {
- if (!ff_vk_mt_is_np_rgb(s->vkctx.input_format)) {
+ const AVPixFmtDescriptor *idesc = av_pix_fmt_desc_get(s->vkctx.input_format);
+ const AVPixFmtDescriptor *odesc = av_pix_fmt_desc_get(s->vkctx.output_format);
+ const int iplanes = av_pix_fmt_count_planes(s->vkctx.input_format);
+ const int oplanes = av_pix_fmt_count_planes(s->vkctx.output_format);
+
+ const int irgb = idesc->flags & AV_PIX_FMT_FLAG_RGB;
+ const int iyuv = !irgb && idesc->nb_components >= 2;
+ const int iplanar = idesc->flags & AV_PIX_FMT_FLAG_PLANAR;
+ const int orgb = odesc->flags & AV_PIX_FMT_FLAG_RGB;
+ const int oyuv = !orgb && idesc->nb_components >= 2;
+ const int oplanar = odesc->flags & AV_PIX_FMT_FLAG_PLANAR;
+
+ if (iyuv && oyuv && iplanar && oplanar && iplanes == oplanes && iplanes > 1) {
+ if (idesc->log2_chroma_w != odesc->log2_chroma_w ||
+ idesc->log2_chroma_h != odesc->log2_chroma_h) {
+ av_log(avctx, AV_LOG_ERROR, "Unsupported input format for conversion\n");
+ return AVERROR(EINVAL);
+ }
+ if (s->out_range != AVCOL_RANGE_UNSPECIFIED) {
+ av_log(avctx, AV_LOG_ERROR, "Cannot change range in yuv2yuv conversion\n");
+ return AVERROR(EINVAL);
+ }
+ if (inlink->w == outlink->w && inlink->h == outlink->h)
+ s->scaler = F_NEAREST;
+ } else if (!ff_vk_mt_is_np_rgb(s->vkctx.input_format)) {
av_log(avctx, AV_LOG_ERROR, "Unsupported input format for conversion\n");
return AVERROR(EINVAL);
}
if (s->vkctx.output_format != AV_PIX_FMT_NV12 &&
+ s->vkctx.output_format != AV_PIX_FMT_P010 &&
+ s->vkctx.output_format != AV_PIX_FMT_P012 &&
+ s->vkctx.output_format != AV_PIX_FMT_P016 &&
+ s->vkctx.output_format != AV_PIX_FMT_NV16 &&
+ s->vkctx.output_format != AV_PIX_FMT_P210 &&
+ s->vkctx.output_format != AV_PIX_FMT_P212 &&
+ s->vkctx.output_format != AV_PIX_FMT_P216 &&
+ s->vkctx.output_format != AV_PIX_FMT_NV24 &&
+ s->vkctx.output_format != AV_PIX_FMT_P410 &&
+ s->vkctx.output_format != AV_PIX_FMT_P412 &&
+ s->vkctx.output_format != AV_PIX_FMT_P416 &&
s->vkctx.output_format != AV_PIX_FMT_YUV420P &&
- s->vkctx.output_format != AV_PIX_FMT_YUV444P) {
+ s->vkctx.output_format != AV_PIX_FMT_YUV420P10 &&
+ s->vkctx.output_format != AV_PIX_FMT_YUV420P12 &&
+ s->vkctx.output_format != AV_PIX_FMT_YUV420P16 &&
+ s->vkctx.output_format != AV_PIX_FMT_YUV422P &&
+ s->vkctx.output_format != AV_PIX_FMT_YUV422P10 &&
+ s->vkctx.output_format != AV_PIX_FMT_YUV422P12 &&
+ s->vkctx.output_format != AV_PIX_FMT_YUV422P16 &&
+ s->vkctx.output_format != AV_PIX_FMT_YUV444P &&
+ s->vkctx.output_format != AV_PIX_FMT_YUV444P10 &&
+ s->vkctx.output_format != AV_PIX_FMT_YUV444P12 &&
+ s->vkctx.output_format != AV_PIX_FMT_YUV444P16) {
av_log(avctx, AV_LOG_ERROR, "Unsupported output format\n");
return AVERROR(EINVAL);
}
Index: FFmpeg/libavfilter/vf_transpose_vulkan.c
===================================================================
--- FFmpeg.orig/libavfilter/vf_transpose_vulkan.c
+++ FFmpeg/libavfilter/vf_transpose_vulkan.c
@@ -49,6 +49,8 @@ static av_cold int init_filter(AVFilterC
TransposeVulkanContext *s = ctx->priv;
FFVulkanContext *vkctx = &s->vkctx;
+ const AVPixFmtDescriptor *pix_desc = av_pix_fmt_desc_get(s->vkctx.output_format);
+ const int is_422 = pix_desc->log2_chroma_w == 1 && !pix_desc->log2_chroma_h;
const int planes = av_pix_fmt_count_planes(s->vkctx.output_format);
FFVulkanShader *shd = &s->shd;
FFVkSPIRVCompiler *spv;
@@ -100,19 +102,24 @@ static av_cold int init_filter(AVFilterC
GLSLC(0, void main() );
GLSLC(0, { );
GLSLC(1, ivec2 size; );
+ GLSLC(1, ivec2 ipos; );
GLSLC(1, ivec2 pos = ivec2(gl_GlobalInvocationID.xy); );
for (int i = 0; i < planes; i++) {
+ int is_422_uv = i && is_422;
GLSLC(0, );
GLSLF(1, size = imageSize(output_images[%i]); ,i);
GLSLC(1, if (IS_WITHIN(pos, size)) { );
if (s->dir == TRANSPOSE_CCLOCK)
- GLSLF(2, vec4 res = imageLoad(input_images[%i], ivec2(size.y - pos.y, pos.x)); ,i);
+ GLSLC(2, ipos = ivec2(size.y - 1 - pos.y, pos.x); );
else if (s->dir == TRANSPOSE_CLOCK_FLIP || s->dir == TRANSPOSE_CLOCK) {
- GLSLF(2, vec4 res = imageLoad(input_images[%i], ivec2(size.yx - pos.yx)); ,i);
+ GLSLC(2, ipos = ivec2(size.yx - 1 - pos.yx); );
if (s->dir == TRANSPOSE_CLOCK)
- GLSLC(2, pos = ivec2(pos.x, size.y - pos.y); );
+ GLSLC(2, pos = ivec2(pos.x, size.y - 1 - pos.y); );
} else
- GLSLF(2, vec4 res = imageLoad(input_images[%i], pos.yx); ,i);
+ GLSLC(2, ipos = pos.yx; );
+ if (is_422_uv)
+ GLSLC(2, ipos = ivec2(ipos.x >> 1, ipos.y << 1); );
+ GLSLF(2, vec4 res = imageLoad(input_images[%i], ipos); ,i);
GLSLF(2, imageStore(output_images[%i], pos, res); ,i);
GLSLC(1, } );
}
@@ -161,11 +168,9 @@ static int filter_frame(AVFilterLink *in
RET(av_frame_copy_props(out, in));
if (in->sample_aspect_ratio.num)
+ out->sample_aspect_ratio = av_inv_q(in->sample_aspect_ratio);
+ else
out->sample_aspect_ratio = in->sample_aspect_ratio;
- else {
- out->sample_aspect_ratio.num = in->sample_aspect_ratio.den;
- out->sample_aspect_ratio.den = in->sample_aspect_ratio.num;
- }
av_frame_free(&in);
@@ -1,43 +0,0 @@
Index: FFmpeg/libavfilter/vf_tonemap_vaapi.c
===================================================================
--- FFmpeg.orig/libavfilter/vf_tonemap_vaapi.c
+++ FFmpeg/libavfilter/vf_tonemap_vaapi.c
@@ -56,7 +56,7 @@ static int tonemap_vaapi_save_metadata(A
AVContentLightMetadata *light_meta;
if (input_frame->color_trc != AVCOL_TRC_SMPTE2084) {
- av_log(avctx, AV_LOG_WARNING, "Only support HDR10 as input for vaapi tone-mapping\n");
+ av_log(avctx, AV_LOG_DEBUG, "Only support HDR10 as input for vaapi tone-mapping\n");
}
ctx->src_display = av_frame_get_side_data(input_frame,
@@ -64,8 +64,7 @@ static int tonemap_vaapi_save_metadata(A
if (ctx->src_display) {
hdr_meta = (AVMasteringDisplayMetadata *)ctx->src_display->data;
if (!hdr_meta) {
- av_log(avctx, AV_LOG_ERROR, "No mastering display data\n");
- return AVERROR(EINVAL);
+ av_log(avctx, AV_LOG_DEBUG, "No mastering display data\n");
}
if (hdr_meta->has_luminance) {
@@ -122,8 +121,7 @@ static int tonemap_vaapi_save_metadata(A
ctx->in_metadata.white_point_y);
}
} else {
- av_log(avctx, AV_LOG_ERROR, "No mastering display data from input\n");
- return AVERROR(EINVAL);
+ av_log(avctx, AV_LOG_DEBUG, "No mastering display data from input\n");
}
ctx->src_light = av_frame_get_side_data(input_frame,
@@ -131,8 +129,7 @@ static int tonemap_vaapi_save_metadata(A
if (ctx->src_light) {
light_meta = (AVContentLightMetadata *)ctx->src_light->data;
if (!light_meta) {
- av_log(avctx, AV_LOG_ERROR, "No light metadata\n");
- return AVERROR(EINVAL);
+ av_log(avctx, AV_LOG_DEBUG, "No light metadata\n");
}
ctx->in_metadata.max_content_light_level = light_meta->MaxCLL;
@@ -0,0 +1,95 @@
Index: FFmpeg/libavcodec/cuviddec.c
===================================================================
--- FFmpeg.orig/libavcodec/cuviddec.c
+++ FFmpeg/libavcodec/cuviddec.c
@@ -164,9 +164,17 @@ static int CUDAAPI cuvid_handle_video_se
avctx->height = cuinfo.display_area.bottom - cuinfo.display_area.top;
}
- // target width/height need to be multiples of two
- cuinfo.ulTargetWidth = avctx->width = (avctx->width + 1) & ~1;
- cuinfo.ulTargetHeight = avctx->height = (avctx->height + 1) & ~1;
+ // NVDEC target dimensions must be even-aligned for internal surface allocation.
+ // For chroma-subsampled formats (420/422), the output dimensions must also be
+ // even. For monochrome/444, keep the original output dimensions and only
+ // even-align the NVDEC target — the frame copy will crop to avctx dimensions.
+ cuinfo.ulTargetWidth = (avctx->width + 1) & ~1;
+ cuinfo.ulTargetHeight = (avctx->height + 1) & ~1;
+ if (format->chroma_format == cudaVideoChromaFormat_420 ||
+ format->chroma_format == cudaVideoChromaFormat_422) {
+ avctx->width = cuinfo.ulTargetWidth;
+ avctx->height = cuinfo.ulTargetHeight;
+ }
// aspect ratio conversion, 1:1, depends on scaled resolution
cuinfo.target_rect.left = 0;
@@ -916,7 +924,7 @@ static av_cold int cuvid_decode_init(AVC
if (probe_desc && probe_desc->nb_components)
probed_bit_depth = probe_desc->comp[0].depth;
- if (probe_desc && !probe_desc->log2_chroma_w && !probe_desc->log2_chroma_h)
+ if (probe_desc && probe_desc->nb_components > 1 && !probe_desc->log2_chroma_w && !probe_desc->log2_chroma_h)
is_yuv444 = 1;
#ifdef NVDEC_HAVE_422_SUPPORT
@@ -1103,7 +1111,7 @@ static av_cold int cuvid_decode_init(AVC
// Skip first 4 bytes of AV1CodecConfigurationRecord to keep configOBUs
// only, otherwise cuvidParseVideoData report unknown error.
if (avctx->codec->id == AV_CODEC_ID_AV1 &&
- extradata_size > 4 &&
+ extradata_size >= 4 &&
extradata[0] & 0x80) {
extradata += 4;
extradata_size -= 4;
Index: FFmpeg/libavcodec/nvdec.c
===================================================================
--- FFmpeg.orig/libavcodec/nvdec.c
+++ FFmpeg/libavcodec/nvdec.c
@@ -732,8 +732,18 @@ int ff_nvdec_frame_params(AVCodecContext
chroma_444 = supports_444 && cuvid_chroma_format == cudaVideoChromaFormat_444;
frames_ctx->format = AV_PIX_FMT_CUDA;
- frames_ctx->width = (avctx->coded_width + 1) & ~1;
- frames_ctx->height = (avctx->coded_height + 1) & ~1;
+ // NVDEC target dimensions must be even-aligned for internal surface allocation.
+ // For chroma-subsampled formats (420/422), the output dimensions must also be
+ // even. For monochrome/444, keep the original output dimensions and only
+ // even-align the NVDEC target — the frame copy will crop to avctx dimensions.
+ if (cuvid_chroma_format == cudaVideoChromaFormat_420 ||
+ cuvid_chroma_format == cudaVideoChromaFormat_422) {
+ frames_ctx->width = (avctx->coded_width + 1) & ~1;
+ frames_ctx->height = (avctx->coded_height + 1) & ~1;
+ } else {
+ frames_ctx->width = avctx->coded_width;
+ frames_ctx->height = avctx->coded_height;
+ }
/*
* We add two extra frames to the pool to account for deinterlacing filters
* holding onto their frames.
Index: FFmpeg/libavcodec/nvenc.c
===================================================================
--- FFmpeg.orig/libavcodec/nvenc.c
+++ FFmpeg/libavcodec/nvenc.c
@@ -683,7 +683,9 @@ static int nvenc_check_capabilities(AVCo
#ifdef NVENC_HAVE_MVHEVC
ctx->multiview_supported = nvenc_check_cap(avctx, NV_ENC_CAPS_SUPPORT_MVHEVC_ENCODE) > 0;
- if(ctx->profile == NV_ENC_HEVC_PROFILE_MULTIVIEW_MAIN && !ctx->multiview_supported) {
+ if (avctx->codec_id == AV_CODEC_ID_HEVC &&
+ ctx->profile == NV_ENC_HEVC_PROFILE_MULTIVIEW_MAIN &&
+ !ctx->multiview_supported) {
av_log(avctx, AV_LOG_WARNING, "Multiview not supported by the device\n");
return AVERROR(ENOSYS);
}
@@ -1359,6 +1361,11 @@ static av_cold int nvenc_setup_h264_conf
case NV_ENC_H264_PROFILE_BASELINE:
cc->profileGUID = NV_ENC_H264_PROFILE_BASELINE_GUID;
avctx->profile = AV_PROFILE_H264_BASELINE;
+ if (cc->frameIntervalP > 1) {
+ av_log(avctx, AV_LOG_WARNING,
+ "B-frames are not supported by H.264 Baseline profile, disabling.\n");
+ cc->frameIntervalP = 1;
+ }
break;
case NV_ENC_H264_PROFILE_MAIN:
cc->profileGUID = NV_ENC_H264_PROFILE_MAIN_GUID;
@@ -0,0 +1,52 @@
Index: FFmpeg/libavformat/mpegts.c
===================================================================
--- FFmpeg.orig/libavformat/mpegts.c
+++ FFmpeg/libavformat/mpegts.c
@@ -2263,6 +2263,7 @@ int ff_parse_mpeg2_descriptor(AVFormatCo
sti->need_parsing = AVSTREAM_PARSE_FULL;
sti->need_context_update = 1;
}
+ break;
}
if (ext_desc_tag == SUPPLEMENTARY_AUDIO_DESCRIPTOR) {
int flags;
@@ -2300,22 +2301,14 @@ int ff_parse_mpeg2_descriptor(AVFormatCo
if (language[0])
av_dict_set(&st->metadata, "language", language, 0);
}
+ break;
}
- break;
- case AC3_DESCRIPTOR:
- {
- int component_type_flag = get8(pp, desc_end) & (1 << 7);
- if (component_type_flag) {
- int component_type = get8(pp, desc_end);
- int service_type_mask = 0x38; // 0b00111000
- int service_type = ((component_type & service_type_mask) >> 3);
- if (service_type == 0x02 /* 0b010 */) {
- st->disposition |= AV_DISPOSITION_DESCRIPTIONS;
- av_log(ts ? ts->stream : fc, AV_LOG_DEBUG, "New track disposition for id %u: %u\n", st->id, st->disposition);
- }
- }
+ if (ext_desc_tag == AC4_DESCRIPTOR) {
+ st->codecpar->codec_id = AV_CODEC_ID_AC4;
+ st->codecpar->codec_type = AVMEDIA_TYPE_AUDIO;
}
break;
+ case AC3_DESCRIPTOR:
case ENHANCED_AC3_DESCRIPTOR:
{
int component_type_flag = get8(pp, desc_end) & (1 << 7);
Index: FFmpeg/libavformat/mpegts.h
===================================================================
--- FFmpeg.orig/libavformat/mpegts.h
+++ FFmpeg/libavformat/mpegts.h
@@ -228,6 +228,7 @@ https://developer.apple.com/library/arch
/* DVB descriptor_tag_extension values from
ETSI EN 300 468 Table 109: Possible locations of extended descriptors */
#define SUPPLEMENTARY_AUDIO_DESCRIPTOR 0x06
+#define AC4_DESCRIPTOR 0x15
/** see "Dolby Vision Streams Within the MPEG-2 Transport Stream Format"
https://professional.dolby.com/siteassets/content-creation/dolby-vision-for-content-creators/dolby-vision-bitstreams-in-mpeg-2-transport-stream-multiplex-v1.2.pdf */
@@ -0,0 +1,21 @@
Index: FFmpeg/libavfilter/vf_libplacebo.c
===================================================================
--- FFmpeg.orig/libavfilter/vf_libplacebo.c
+++ FFmpeg/libavfilter/vf_libplacebo.c
@@ -1478,6 +1478,16 @@ static int libplacebo_config_output(AVFi
AVHWFramesContext *hwfc;
AVVulkanFramesContext *vkfc;
+ if (!((s->color_primaries == -1 &&
+ s->color_trc == -1) ||
+ (s->color_primaries == AVCOL_PRI_BT2020 &&
+ s->color_trc == AVCOL_TRC_SMPTE2084))) {
+ av_frame_side_data_remove(&outlink->side_data, &outlink->nb_side_data,
+ AV_FRAME_DATA_MASTERING_DISPLAY_METADATA);
+ av_frame_side_data_remove(&outlink->side_data, &outlink->nb_side_data,
+ AV_FRAME_DATA_CONTENT_LIGHT_LEVEL);
+ }
+
/* Frame dimensions */
RET(ff_scale_eval_dimensions(s, s->w_expr, s->h_expr, inlink, outlink,
&outlink->w, &outlink->h));
@@ -1,278 +0,0 @@
Index: FFmpeg/libavcodec/ac3_parser.c
===================================================================
--- FFmpeg.orig/libavcodec/ac3_parser.c
+++ FFmpeg/libavcodec/ac3_parser.c
@@ -73,6 +73,201 @@ int ff_ac3_find_syncword(const uint8_t *
return i;
}
+#ifndef EAC3_SR_CODE_REDUCED
+ #define EAC3_SR_CODE_REDUCED 3
+#endif
+
+// see also ff_eac3_parse_header()
+static int eac3_parse_addbsi_from_header(GetBitContext *gbc,
+ AC3HeaderInfo *hdr,
+ uint8_t *flag_ec3_extension_type_a,
+ uint8_t *complexity_index_type_a)
+{
+ int i, blk;
+
+ if (!gbc || !hdr || !flag_ec3_extension_type_a || !complexity_index_type_a)
+ return AVERROR(EINVAL);
+
+ if (hdr->frame_type == EAC3_FRAME_TYPE_RESERVED)
+ return AC3_PARSE_ERROR_FRAME_TYPE;
+ if (hdr->substreamid)
+ return AC3_PARSE_ERROR_FRAME_TYPE;
+ if (hdr->sr_code == EAC3_SR_CODE_REDUCED)
+ return AVERROR_PATCHWELCOME;
+
+ skip_bits(gbc, 5); // skip bitstream id
+
+ /* volume control params */
+ for (i = 0; i < (hdr->channel_mode ? 1 : 2); i++) {
+ skip_bits(gbc, 5); // skip dialog_normalization[i]
+ if (get_bits1(gbc)) {
+ skip_bits(gbc, 8); // skip heavy_dynamic_range[i]
+ }
+ }
+
+ /* dependent stream channel map */
+ if (hdr->frame_type == EAC3_FRAME_TYPE_DEPENDENT) {
+ if (get_bits1(gbc)) {
+ skip_bits(gbc, 16); // skip channel_map
+ }
+ }
+
+ /* mixing metadata */
+ if (get_bits1(gbc)) {
+ /* center and surround mix levels */
+ if (hdr->channel_mode > AC3_CHMODE_STEREO) {
+ skip_bits(gbc, 2); // skip preferred_downmix
+ if (hdr->channel_mode & 1) {
+ /* if three front channels exist */
+ skip_bits(gbc, 3); // skip center_mix_level_ltrt
+ skip_bits(gbc, 3); // skip center_mix_level
+ }
+ if (hdr->channel_mode & 4) {
+ /* if a surround channel exists */
+ skip_bits(gbc, 3); // skip surround_mix_level_ltrt
+ skip_bits(gbc, 3); // skip surround_mix_level
+ }
+ }
+
+ /* lfe mix level */
+ if (hdr->lfe_on && get_bits1(gbc)) {
+ skip_bits(gbc, 5); // skip lfe_mix_level
+ }
+
+ /* info for mixing with other streams and substreams */
+ if (hdr->frame_type == EAC3_FRAME_TYPE_INDEPENDENT) {
+ for (i = 0; i < (hdr->channel_mode ? 1 : 2); i++) {
+ // TODO: apply program scale factor
+ if (get_bits1(gbc)) {
+ skip_bits(gbc, 6); // skip program scale factor
+ }
+ }
+ if (get_bits1(gbc)) {
+ skip_bits(gbc, 6); // skip external program scale factor
+ }
+ /* skip mixing parameter data */
+ switch(get_bits(gbc, 2)) {
+ case 1: skip_bits(gbc, 5); break;
+ case 2: skip_bits(gbc, 12); break;
+ case 3: {
+ int mix_data_size = (get_bits(gbc, 5) + 2) << 3;
+ skip_bits_long(gbc, mix_data_size);
+ break;
+ }
+ }
+ /* skip pan information for mono or dual mono source */
+ if (hdr->channel_mode < AC3_CHMODE_STEREO) {
+ for (i = 0; i < (hdr->channel_mode ? 1 : 2); i++) {
+ if (get_bits1(gbc)) {
+ /* note: this is not in the ATSC A/52B specification
+ reference: ETSI TS 102 366 V1.1.1
+ section: E.1.3.1.25 */
+ skip_bits(gbc, 8); // skip pan mean direction index
+ skip_bits(gbc, 6); // skip reserved paninfo bits
+ }
+ }
+ }
+ /* skip mixing configuration information */
+ if (get_bits1(gbc)) {
+ for (blk = 0; blk < hdr->num_blocks; blk++) {
+ if (hdr->num_blocks == 1 || get_bits1(gbc)) {
+ skip_bits(gbc, 5);
+ }
+ }
+ }
+ }
+ }
+
+ /* informational metadata */
+ if (get_bits1(gbc)) {
+ skip_bits(gbc, 3); // skip bitstream_mode
+ skip_bits(gbc, 2); // skip copyright bit and original bitstream bit
+ if (hdr->channel_mode == AC3_CHMODE_STEREO) {
+ skip_bits(gbc, 2); // skip dolby_surround_mode
+ skip_bits(gbc, 2); // skip dolby_headphone_mode
+ }
+ if (hdr->channel_mode >= AC3_CHMODE_2F2R) {
+ skip_bits(gbc, 2); // skip dolby_surround_ex_mode
+ }
+ for (i = 0; i < (hdr->channel_mode ? 1 : 2); i++) {
+ if (get_bits1(gbc)) {
+ skip_bits(gbc, 8); // skip mix level, room type, and A/D converter type
+ }
+ }
+ if (hdr->sr_code != EAC3_SR_CODE_REDUCED) {
+ skip_bits1(gbc); // skip source sample rate code
+ }
+ }
+
+ /* converter synchronization flag
+ If frames are less than six blocks, this bit should be turned on
+ once every 6 blocks to indicate the start of a frame set.
+ reference: RFC 4598, Section 2.1.3 Frame Sets */
+ if (hdr->frame_type == EAC3_FRAME_TYPE_INDEPENDENT && hdr->num_blocks != 6) {
+ skip_bits1(gbc); // skip converter synchronization flag
+ }
+
+ /* original frame size code if this stream was converted from AC-3 */
+ if (hdr->frame_type == EAC3_FRAME_TYPE_AC3_CONVERT &&
+ (hdr->num_blocks == 6 || get_bits1(gbc))) {
+ skip_bits(gbc, 6); // skip frame size code
+ }
+
+ /* additional bitstream info */
+ if (get_bits1(gbc)) {
+ int addbsil = get_bits(gbc, 6);
+ for (i = 0; i < addbsil + 1; i++) {
+ if (i == 0) {
+ /* In this 8 bit chunk, the LSB is equal to flag_ec3_extension_type_a
+ which can be used to detect Atmos presence */
+ skip_bits(gbc, 7);
+ if (get_bits1(gbc)) {
+ *flag_ec3_extension_type_a = 1;
+ *complexity_index_type_a = get_bits(gbc, 8);
+ /* The upper limit of complexity_index_type_a is 16 */
+ *complexity_index_type_a = FFMIN(*complexity_index_type_a, 16);
+ ++i;
+ }
+ } else {
+ skip_bits(gbc, 8); // skip additional bit stream info
+ }
+ }
+ }
+ return 0;
+}
+
+int avpriv_ac3_parse_header2(AC3HeaderInfo **phdr,
+ const uint8_t *buf, size_t size,
+ uint8_t *flag_ec3_extension_type_a,
+ uint8_t *complexity_index_type_a)
+{
+ GetBitContext gb;
+ AC3HeaderInfo *hdr;
+ int err;
+
+ if (!*phdr)
+ *phdr = av_mallocz(sizeof(AC3HeaderInfo));
+ if (!*phdr)
+ return AVERROR(ENOMEM);
+ hdr = *phdr;
+
+ err = init_get_bits8(&gb, buf, size);
+ if (err < 0)
+ return AVERROR_INVALIDDATA;
+ err = ff_ac3_parse_header(&gb, hdr);
+ if (err < 0)
+ return AVERROR_INVALIDDATA;
+
+ // eac3 addbsi
+ if (!(hdr->bitstream_id <= 10)) {
+ eac3_parse_addbsi_from_header(&gb, hdr,
+ flag_ec3_extension_type_a,
+ complexity_index_type_a);
+ }
+
+ return get_bits_count(&gb);
+}
+
int ff_ac3_parse_header(GetBitContext *gbc, AC3HeaderInfo *hdr)
{
int frame_size_code;
Index: FFmpeg/libavcodec/ac3_parser_internal.h
===================================================================
--- FFmpeg.orig/libavcodec/ac3_parser_internal.h
+++ FFmpeg/libavcodec/ac3_parser_internal.h
@@ -86,6 +86,11 @@ int ff_ac3_parse_header(GetBitContext *g
int avpriv_ac3_parse_header(AC3HeaderInfo **hdr, const uint8_t *buf,
size_t size);
+int avpriv_ac3_parse_header2(AC3HeaderInfo **phdr,
+ const uint8_t *buf, size_t size,
+ uint8_t *flag_ec3_extension_type_a,
+ uint8_t *complexity_index_type_a);
+
int ff_ac3_find_syncword(const uint8_t *buf, int buf_size);
#endif /* AVCODEC_AC3_PARSER_INTERNAL_H */
Index: FFmpeg/libavformat/movenc.c
===================================================================
--- FFmpeg.orig/libavformat/movenc.c
+++ FFmpeg/libavformat/movenc.c
@@ -393,6 +393,9 @@ struct eac3_info {
uint16_t chan_loc;
/* if there is no dependent substream, then one bit reserved instead */
} substream[1]; /* TODO: support 8 independent substreams */
+ /* addbsi */
+ uint8_t flag_ec3_extension_type_a; /* indicates the enhanced AC-3 extension, 1 bit */
+ uint8_t complexity_index_type_a; /* indicates the decoding complexity, 8 bits */
};
static int mov_write_ac3_tag(AVFormatContext *s, AVIOContext *pb, MOVTrack *track)
@@ -458,7 +461,9 @@ static int handle_eac3(MOVMuxContext *mo
if (!info->pkt && !(info->pkt = av_packet_alloc()))
return AVERROR(ENOMEM);
- if ((ret = avpriv_ac3_parse_header(&hdr, pkt->data, pkt->size)) < 0) {
+ if ((ret = avpriv_ac3_parse_header2(&hdr, pkt->data, pkt->size,
+ &info->flag_ec3_extension_type_a,
+ &info->complexity_index_type_a)) < 0) {
if (ret == AVERROR(ENOMEM))
goto end;
@@ -534,7 +539,8 @@ static int handle_eac3(MOVMuxContext *mo
while (cumul_size != pkt->size) {
GetBitContext gbc;
int i;
- ret = avpriv_ac3_parse_header(&hdr, pkt->data + cumul_size, pkt->size - cumul_size);
+ ret = avpriv_ac3_parse_header2(&hdr, pkt->data + cumul_size, pkt->size - cumul_size,
+ &info->flag_ec3_extension_type_a, &info->complexity_index_type_a);
if (ret < 0)
goto end;
if (hdr->frame_type != EAC3_FRAME_TYPE_DEPENDENT) {
@@ -614,7 +620,7 @@ static int mov_write_eac3_tag(AVFormatCo
}
info = track->eac3_priv;
- size = 2 + ((32 * (info->num_ind_sub + 1) + 7) >> 3);
+ size = 2 + ((32 * (info->num_ind_sub + 1) + 7 + 1 + 8) >> 3);
buf = av_malloc(size);
if (!buf) {
return AVERROR(ENOMEM);
@@ -639,6 +645,11 @@ static int mov_write_eac3_tag(AVFormatCo
put_bits(&pbc, 9, info->substream[i].chan_loc);
}
}
+ if (info->flag_ec3_extension_type_a == 1) {
+ put_bits(&pbc, 7, 0); /* reserved */
+ put_bits(&pbc, 1, info->flag_ec3_extension_type_a);
+ put_bits(&pbc, 8, info->complexity_index_type_a);
+ }
flush_put_bits(&pbc);
size = put_bytes_output(&pbc);
@@ -0,0 +1,13 @@
Index: FFmpeg/libavformat/movenc.c
===================================================================
--- FFmpeg.orig/libavformat/movenc.c
+++ FFmpeg/libavformat/movenc.c
@@ -6929,7 +6929,7 @@ int ff_mov_write_packet(AVFormatContext
} else if (par->codec_id == AV_CODEC_ID_HEVC && trk->extradata_size[trk->last_stsd_index] > 6 &&
(AV_RB24(trk->extradata[trk->last_stsd_index]) == 1 || AV_RB32(trk->extradata[trk->last_stsd_index]) == 1)) {
/* extradata is Annex B, assume the bitstream is too and convert it */
- int filter_ps = (trk->tag == MKTAG('h','v','c','1'));
+ int filter_ps = 0; // Always disable it as a Wa for AMD hevc_vaapi encoder
if (trk->hint_track >= 0 && trk->hint_track < mov->nb_tracks) {
ret = ff_hevc_annexb2mp4_buf(pkt->data, &reformatted_data,
&size, filter_ps, NULL);

Some files were not shown because too many files have changed in this diff Show More