This is an automated email from the git hooks/post-receive script. Git pushed a commit to branch master in repository ffmpeg.
commit 00d08d1955f036571830909d0af6c34872f4bfe6 Author: Niklas Haas <[email protected]> AuthorDate: Fri Jul 17 12:51:54 2026 +0200 Commit: Niklas Haas <[email protected]> CommitDate: Fri Oct 2 14:13:16 2026 +0200 swscale/x86/ops_int: implement integer linear transformations This is a bit inefficient for the hyper-special case of e.g. a single integer multiplication only. In particular, the AVX2 path actually ends up slower than the SSE4 path; though we still need to implement it due to the block size needing to match the rest of the chain. I plan on maybe adding a special case to cover the isolated / single component case down the line. checkasm: - CPU: AMD Ryzen 9 9950X3D 16-Core Processor (00B40F40) - Timing source: x86 (rdtsc) Benchmark results: name cycles (vs ref) u8_linear_x_x000x_c: 1039.5 u8_linear_x_x000x_x86_sse4: 428.5 ( 2.41x) u8_linear_x_x000x_x86_avx2: 907.4 ( 1.15x) u8_linear_xyz_x000x_0x00x_00x0x_c: 1177.7 u8_linear_xyz_x000x_0x00x_00x0x_x86_sse4: 714.1 ( 1.65x) u8_linear_xyz_x000x_0x00x_00x0x_x86_avx2: 843.1 ( 1.39x) u8_linear_xyz_x0000_0x000_00x00_c: 1131.5 u8_linear_xyz_x0000_0x000_00x00_x86_sse4: 713.6 ( 1.58x) u8_linear_xyz_x0000_0x000_00x00_x86_avx2: 867.3 ( 1.30x) u8_linear_y_0x000_c: 1031.3 u8_linear_y_0x000_x86_sse4: 389.2 ( 2.65x) u8_linear_y_0x000_x86_avx2: 911.5 ( 1.13x) u16_linear_x_x000x_c: 1206.4 u16_linear_x_x000x_x86_avx2: 610.4 ( 1.98x) u16_linear_xyz_x000x_0x00x_00x0x_c: 1445.1 u16_linear_xyz_x000x_0x00x_00x0x_x86_avx2: 503.6 ( 2.87x) u16_linear_xyz_x0000_0x000_00x00_c: 1324.8 u16_linear_xyz_x0000_0x000_00x00_x86_avx2: 506.2 ( 2.62x) u16_linear_xyzw_x0000_0x000_00x00_000x0_c: 1424.6 u16_linear_xyzw_x0000_0x000_00x00_000x0_x86_avx2: 467.0 ( 3.05x) u32_linear_x_x000x_c: 1762.4 u32_linear_x_x000x_x86_avx2: 855.9 ( 2.06x) u32_linear_xyz_x000x_0x00x_00x0x_c: 2461.8 u32_linear_xyz_x000x_0x00x_00x0x_x86_avx2: 919.9 ( 2.68x) u32_linear_xyz_x0000_0x000_00x00_c: 2297.8 u32_linear_xyz_x0000_0x000_00x00_x86_avx2: 865.6 ( 2.65x) Sponsored-by: Sovereign Tech Fund Signed-off-by: Niklas Haas <[email protected]> --- libswscale/x86/ops.c | 44 ++++++++++++++++- libswscale/x86/ops_float.asm | 12 ++--- libswscale/x86/ops_include.asm | 11 ++++- libswscale/x86/ops_int.asm | 108 +++++++++++++++++++++++++++++++++++++++++ 4 files changed, 165 insertions(+), 10 deletions(-) diff --git a/libswscale/x86/ops.c b/libswscale/x86/ops.c index 00994e4d64..e79b1cf117 100644 --- a/libswscale/x86/ops.c +++ b/libswscale/x86/ops.c @@ -277,12 +277,52 @@ static int setup_dither(const SwsImplParams *params, SwsImplResult *out) return 0; } +static void splat_lane(void *dst, SwsPixelType type, SwsPixel px) +{ + switch (ff_sws_pixel_type_size(type)) { + case 1: + memset(dst, px.u8, 16); + break; + case 2: + for (int i = 0; i < 8; i++) + ((uint16_t *) dst)[i] = px.u16; + break; + case 4: + for (int i = 0; i < 4; i++) + ((uint32_t *) dst)[i] = px.u32; + break; + } +} + static int setup_linear(const SwsImplParams *params, SwsImplResult *out) { const SwsUOp *uop = params->uop; - out->priv.ptr = av_memdup(uop->data.mat4, sizeof(uop->data.mat4)); + if (uop->type == SWS_PIXEL_F32) { + out->priv.ptr = av_memdup(uop->data.mat4, sizeof(uop->data.mat4)); + out->free = ff_op_priv_free; + return out->priv.ptr ? 0 : AVERROR(ENOMEM); + } + + uint8_t *mat = av_malloc(4 * 5 * 16); /* one lane per component */ + if (!mat) + return AVERROR(ENOMEM); + out->priv.ptr = mat; out->free = ff_op_priv_free; - return out->priv.ptr ? 0 : AVERROR(ENOMEM); + + for (int i = 0; i < 4; i++) { + for (int j = 0; j < 5; j++) { + SwsPixel px = uop->data.mat4[i][j]; + SwsPixelType type = uop->type; + if (type == SWS_PIXEL_U8) { + type = SWS_PIXEL_U16; /* for pmullw */ + px.u16 = (px.u8 << 8) | px.u8; + } + + splat_lane(mat, type, px); + mat += 16; + } + } + return 0; } static bool uop_is_type_invariant(const SwsUOpType uop) diff --git a/libswscale/x86/ops_float.asm b/libswscale/x86/ops_float.asm index 3b8055c31a..2c1858c14a 100644 --- a/libswscale/x86/ops_float.asm +++ b/libswscale/x86/ops_float.asm @@ -534,9 +534,7 @@ IF W, maxps mw2, m11 ;--------------------------------------------------------- ; Linear operations -%define LIN_MASK(I, J) (1 << (5 * (I) + (J))) - -%macro linear_muladd 5 ; dst, src, use_coef, coef, use_fma +%macro linear_muladdps 5 ; dst, src, use_coef, coef, use_fma %if INIT ; dst is already initialized %if %3 && %5 fmaddps %1, %4, %2, %1 @@ -570,10 +568,10 @@ IF LOAD(0), vbroadcastss m12, [%2 + 0 * BYTES] IF LOAD(1), vbroadcastss m13, [%2 + 1 * BYTES] IF LOAD(2), vbroadcastss m14, [%2 + 2 * BYTES] IF LOAD(3), vbroadcastss m15, [%2 + 3 * BYTES] -IF NEED(0), linear_muladd %1, mx%4, LOAD(0), m12, FMA(0) -IF NEED(1), linear_muladd %1, my%4, LOAD(1), m13, FMA(1) -IF NEED(2), linear_muladd %1, mz%4, LOAD(2), m14, FMA(2) -IF NEED(3), linear_muladd %1, mw%4, LOAD(3), m15, FMA(3) +IF NEED(0), linear_muladdps %1, mx%4, LOAD(0), m12, FMA(0) +IF NEED(1), linear_muladdps %1, my%4, LOAD(1), m13, FMA(1) +IF NEED(2), linear_muladdps %1, mz%4, LOAD(2), m14, FMA(2) +IF NEED(3), linear_muladdps %1, mw%4, LOAD(3), m15, FMA(3) assert INIT, SWS_UOP_LINEAR should not contain empty rows %endmacro diff --git a/libswscale/x86/ops_include.asm b/libswscale/x86/ops_include.asm index 073ed31e57..85777d7529 100644 --- a/libswscale/x86/ops_include.asm +++ b/libswscale/x86/ops_include.asm @@ -146,6 +146,9 @@ endstruc %define SWS_COMP_INV(mask) ((mask) ^ SWS_COMP_ALL) %define SWS_COMP_ELEMS(N) ((1 << (N)) - 1) +%define LIN_MASK(I, J) (1 << (5 * (I) + (J))) +%define LIN_COL(J) (LIN_MASK(0, J) | LIN_MASK(1, J) | LIN_MASK(2, J) | LIN_MASK(3, J)) + ;--------------------------------------------------------- ; Common macros for declaring operations @@ -326,13 +329,19 @@ endstruc %endif %endmacro -; Alternate name; for nested usage (to work around NASM limitations) +; Alternate names; for nested usage (to work around NASM limitations) %macro IF1 2+ %if %1 %2 %endif %endmacro +%macro IF2 2+ + %if %1 + %2 + %endif +%endmacro + %macro shl_log2 2 ; dst, amount %if %2 == 64 shl %1, 6 diff --git a/libswscale/x86/ops_int.asm b/libswscale/x86/ops_int.asm index 76e369d1fa..6f49ffeafe 100644 --- a/libswscale/x86/ops_int.asm +++ b/libswscale/x86/ops_int.asm @@ -770,6 +770,113 @@ assert 0, SWS_UOP_LINEAR_FMA is not implemented for integer types assert 0, SWS_UOP_DITHER is not implemented for integer types %endmacro +;--------------------------------------------------------- +; Linear operations + +%macro linear_muladdw 4 ; dst, src, use_coef, coef + %if BITS == 32 + %xdefine MUL pmulld + %xdefine ADD paddd + %else + %xdefine MUL pmullw + %xdefine ADD paddw + %endif + %if INIT ; dst is already initialized + %if %3 + MUL %4, %2 + ADD %1, %4 + %else + ADD %1, %2 + %endif + %else + %assign INIT 1 + %if %3 + MUL %1, %2, %4 + %else + mova %1, %2 + %endif + %endif +%endmacro + +%macro linear_row 3 ; dst, src, row +%xdefine NEED(J) (!(ZERO_MASK & LIN_MASK(%3, J))) +%xdefine LOAD(J) (NEED(J) && !(ONE_MASK & LIN_MASK(%3, J))) +%assign INIT 0 ; track whether `dst` already contains data + + %if !(ZERO_MASK & LIN_MASK(%3, 4)) ; nonzero output offset + %assign INIT 1 + VBROADCASTI128 %1, [%2 + 4 * 16] + %endif +IF LOAD(0), VBROADCASTI128 m12, [%2 + 0 * 16] +IF LOAD(1), VBROADCASTI128 m13, [%2 + 1 * 16] +IF LOAD(2), VBROADCASTI128 m14, [%2 + 2 * 16] +IF LOAD(3), VBROADCASTI128 m15, [%2 + 3 * 16] +IF NEED(0), linear_muladdw %1, IN0, LOAD(0), m12 +IF NEED(1), linear_muladdw %1, IN1, LOAD(1), m13 +IF NEED(2), linear_muladdw %1, IN2, LOAD(2), m14 +IF NEED(3), linear_muladdw %1, IN3, LOAD(3), m15 + assert INIT, SWS_UOP_LINEAR should not contain empty rows +%endmacro + +; Swap the high and low bytes of `dst` and `out` and merge back into `dst` +%macro linear_rot 4 ; have_out, need_in, dst, out + %if %1 || %2 ; we also need to rotate pure input registers + %if %1 + psllw %4, 8 + %else + psllw %4, %3, 8 + %endif + psrlw %3, 8 + por %3, %4 + %endif +%endmacro + +%macro linear_pass 0-1 ; suffix +%xdefine USED(J) (!(ZERO_MASK & LIN_COL(J))) +%xdefine IN0 mx%1 +%xdefine IN1 my%1 +%xdefine IN2 mz%1 +%xdefine IN3 mw%1 + +IF1 X, linear_row m8, tmp0q + 0 * 16, 0 +IF1 Y, linear_row m9, tmp0q + 5 * 16, 1 +IF1 Z, linear_row m10, tmp0q + 10 * 16, 2 +IF1 W, linear_row m11, tmp0q + 15 * 16, 3 + + %if BITS == 8 + ; swap high/low bits and compute the other half; this discards the + ; garbage high byte produced by each sub-pass + linear_rot X, USED(0), IN0, m8 + linear_rot Y, USED(1), IN1, m9 + linear_rot Z, USED(2), IN2, m10 + linear_rot W, USED(3), IN3, m11 +IF1 X, linear_row m8, tmp0q + 0 * 16, 0 +IF1 Y, linear_row m9, tmp0q + 5 * 16, 1 +IF1 Z, linear_row m10, tmp0q + 10 * 16, 2 +IF1 W, linear_row m11, tmp0q + 15 * 16, 3 + linear_rot X, USED(0), IN0, m8 + linear_rot Y, USED(1), IN1, m9 + linear_rot Z, USED(2), IN2, m10 + linear_rot W, USED(3), IN3, m11 + %else +IF X, mova IN0, m8 +IF Y, mova IN1, m9 +IF Z, mova IN2, m10 +IF W, mova IN3, m11 + %endif +%endmacro + +%macro LINEAR 2 +%assign ONE_MASK %1 +%assign ZERO_MASK %2 + + mov tmp0q, [implq + SwsOpImpl.priv] ; address of matrix + LOAD_CONT tmp1q + linear_pass +IF2 V2, linear_pass 2 + CONTINUE tmp1q +%endmacro + ;--------------------------------------------------------- ; Instantiate above macros to generate all uop kernels @@ -795,6 +902,7 @@ assert 0, SWS_UOP_DITHER is not implemented for integer types DECL_%1_RSHIFT (RSHIFT) DECL_%1_LINEAR_FMA (LINEAR_FMA) DECL_%1_DITHER (DITHER) + DECL_%1_LINEAR (LINEAR) %endmacro %macro decl_type_invariant 0 -- To stop receiving notification emails like this one, please contact [email protected]. _______________________________________________ ffmpeg-cvslog mailing list -- [email protected] To unsubscribe send an email to [email protected]
