This is an automated email from the git hooks/post-receive script. Git pushed a commit to branch master in repository ffmpeg.
commit b91a82d6dd0e7c32caf34b345130c8fb1309a597 Author: jinbo <[email protected]> AuthorDate: Tue Sep 1 11:28:33 2026 +0800 Commit: Zhao Zhili <[email protected]> CommitDate: Fri Oct 9 10:24:34 2026 +0000 swscale/loongarch: add LSX fast bilinear horizontal scale Add LoongArch LSX optimized implementations of the fast bilinear horizontal scaler (SWS_FAST_BILINEAR, hyscale_fast/hcscale_fast). The vectorized loop processes 8 destination pixels per iteration, computing the source offsets and interpolation weights in registers and gathering the input samples with vshuf_b over a 32-byte window. It covers scaling ratios down to 4x downscaling; beyond that the scalar loop is used. checkasm --bench on 3A5000 4 cores 2.5GHz: hcscale_fast_c: 99.7 hcscale_fast_lsx: 28.1 ( 3.55x) hyscale_fast_c: 51.1 hyscale_fast_lsx: 18.3 ( 2.78x) Performance with: $ ./ffmpeg -cpuflags lsx -f lavfi -i "smptebars=size=3840x2160:rate=30:duration=100,format=nv12" \ -threads 1 -vf "scale=1920:1080:sws_flags=fast_bilinear,format=bgra" -f null - before: 36fps after : 86fps --- libswscale/loongarch/Makefile | 1 + libswscale/loongarch/hscale_fast_bilinear_lsx.c | 162 ++++++++++++++++++++++++ libswscale/loongarch/swscale_init_loongarch.c | 6 + libswscale/loongarch/swscale_loongarch.h | 7 + tests/checkasm/sw_scale.c | 120 ++++++++++++++++++ 5 files changed, 296 insertions(+) diff --git a/libswscale/loongarch/Makefile b/libswscale/loongarch/Makefile index 06aed9d245..0b9c87100e 100644 --- a/libswscale/loongarch/Makefile +++ b/libswscale/loongarch/Makefile @@ -6,6 +6,7 @@ LASX-OBJS-$(CONFIG_SWSCALE) += loongarch/swscale_lasx.o \ loongarch/output_lasx.o LSX-OBJS-$(CONFIG_SWSCALE) += loongarch/swscale.o \ loongarch/swscale_unscaled.o \ + loongarch/hscale_fast_bilinear_lsx.o \ loongarch/swscale_lsx.o \ loongarch/input.o \ loongarch/output.o \ diff --git a/libswscale/loongarch/hscale_fast_bilinear_lsx.c b/libswscale/loongarch/hscale_fast_bilinear_lsx.c new file mode 100644 index 0000000000..7896542251 --- /dev/null +++ b/libswscale/loongarch/hscale_fast_bilinear_lsx.c @@ -0,0 +1,162 @@ +/* + * Copyright (C) 2026 Loongson Technology Co. Ltd. + * Contributed by Bo Jin([email protected]) + * + * This file is part of FFmpeg. + * + * FFmpeg is free software; you can redistribute it and/or + * modify it under the terms of the GNU Lesser General Public + * License as published by the Free Software Foundation; either + * version 2.1 of the License, or (at your option) any later version. + * + * FFmpeg is distributed in the hope that it will be useful, + * but WITHOUT ANY WARRANTY; without even the implied warranty of + * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU + * Lesser General Public License for more details. + * + * You should have received a copy of the GNU Lesser General Public + * License along with FFmpeg; if not, write to the Free Software + * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA + */ + +#include "swscale_loongarch.h" +#include "libavutil/loongarch/loongson_intrinsics.h" + +/* Each vector iteration scales 8 destination pixels. Their source + * position offsets ((xpos & 0xFFFF) + j*xInc) >> 16, j in [0, 7], must + * stay within the 32-byte gather window, which holds while + * xInc <= (1 << 18); otherwise fall back to the scalar loop. */ +#define LSX_HSCALE_FAST_MAX_XINC (1 << 18) + +void ff_hyscale_fast_lsx(SwsInternal *c, int16_t *dst, int dstWidth, + const uint8_t *src, int srcW, int xInc) +{ + int i = 0; + unsigned int xpos = 0; + + if (xInc <= LSX_HSCALE_FAST_MAX_XINC) { + static const int32_t idx32[4] = {0, 1, 2, 3}; + static const int16_t idx16[8] = {0, 1, 2, 3, 4, 5, 6, 7}; + static const uint8_t shuf8[16] = { + 0, 4, 8, 12, 16, 20, 24, 28, + 0x30, 0x30, 0x30, 0x30, 0x30, 0x30, 0x30, 0x30 + }; + + /* [0, xInc, 2*xInc, 3*xInc] as 32-bit lanes */ + __m128i vadd_w = __lsx_vmul_w(__lsx_vreplgr2vr_w(xInc), + __lsx_vld(idx32, 0)); + /* [0, xInc, ..., 7*xInc] modulo 2^16 as 16-bit lanes */ + __m128i vadd16 = __lsx_vmul_h(__lsx_vreplgr2vr_h(xInc), + __lsx_vld(idx16, 0)); + __m128i vx4 = __lsx_vreplgr2vr_w(4 * xInc); + __m128i vshuf8 = __lsx_vld(shuf8, 0); + __m128i v128 = __lsx_vreplgr2vr_h(128); + + for (; i + 8 <= dstWidth && (xpos >> 16) + 32 < srcW; i += 8, xpos += 8 * xInc) { + unsigned int lo = xpos & 0xFFFF; + unsigned int xx = xpos >> 16; + + /* full 32-bit positions of the 8 pixels */ + __m128i vc0 = __lsx_vadd_w(__lsx_vreplgr2vr_w(lo), vadd_w); + __m128i vc1 = __lsx_vadd_w(vc0, vx4); + + /* source offsets (j = 0..7), each in [0, 28] */ + __m128i vperm = __lsx_vshuf_b(__lsx_vsrli_w(vc1, 16), + __lsx_vsrli_w(vc0, 16), vshuf8); + + /* xalpha = (xpos & 0xFFFF) >> 9, 16-bit lanes */ + __m128i valpha = __lsx_vsrli_h(__lsx_vadd_h(__lsx_vreplgr2vr_h(lo), + vadd16), 9); + + __m128i v0 = __lsx_vsllwil_hu_bu(__lsx_vshuf_b(__lsx_vld(src + xx + 16, 0), + __lsx_vld(src + xx, 0), + vperm), 0); + __m128i v1 = __lsx_vsllwil_hu_bu(__lsx_vshuf_b(__lsx_vld(src + xx + 17, 0), + __lsx_vld(src + xx + 1, 0), + vperm), 0); + + __m128i w0 = __lsx_vsub_h(v128, valpha); + __lsx_vst(__lsx_vadd_h(__lsx_vmul_h(v0, w0), __lsx_vmul_h(v1, valpha)), + dst + i, 0); + } + } + + for (; i < dstWidth; i++) { + unsigned int xx = xpos >> 16; + unsigned int xalpha = (xpos & 0xFFFF) >> 9; + dst[i] = (src[xx] << 7) + (src[xx + 1] - src[xx]) * xalpha; + xpos += xInc; + } + for (i = dstWidth - 1; (i * (int64_t)xInc) >> 16 >= srcW - 1; i--) + dst[i] = src[srcW - 1] * 128; +} + +void ff_hcscale_fast_lsx(SwsInternal *c, int16_t *dst1, int16_t *dst2, + int dstWidth, const uint8_t *src1, + const uint8_t *src2, int srcW, int xInc) +{ + int i = 0; + unsigned int xpos = 0; + + if (xInc <= LSX_HSCALE_FAST_MAX_XINC) { + static const int32_t idx32[4] = {0, 1, 2, 3}; + static const int16_t idx16[8] = {0, 1, 2, 3, 4, 5, 6, 7}; + static const uint8_t shuf8[16] = { + 0, 4, 8, 12, 16, 20, 24, 28, + 0x30, 0x30, 0x30, 0x30, 0x30, 0x30, 0x30, 0x30 + }; + + __m128i vadd_w = __lsx_vmul_w(__lsx_vreplgr2vr_w(xInc), + __lsx_vld(idx32, 0)); + __m128i vadd16 = __lsx_vmul_h(__lsx_vreplgr2vr_h(xInc), + __lsx_vld(idx16, 0)); + __m128i vx4 = __lsx_vreplgr2vr_w(4 * xInc); + __m128i vshuf8 = __lsx_vld(shuf8, 0); + __m128i v127 = __lsx_vreplgr2vr_h(127); + + for (; i + 8 <= dstWidth && (xpos >> 16) + 32 < srcW; i += 8, xpos += 8 * xInc) { + unsigned int lo = xpos & 0xFFFF; + unsigned int xx = xpos >> 16; + + __m128i vc0 = __lsx_vadd_w(__lsx_vreplgr2vr_w(lo), vadd_w); + __m128i vc1 = __lsx_vadd_w(vc0, vx4); + + __m128i vperm = __lsx_vshuf_b(__lsx_vsrli_w(vc1, 16), + __lsx_vsrli_w(vc0, 16), vshuf8); + + __m128i valpha = __lsx_vsrli_h(__lsx_vadd_h(__lsx_vreplgr2vr_h(lo), + vadd16), 9); + + __m128i v10 = __lsx_vsllwil_hu_bu(__lsx_vshuf_b(__lsx_vld(src1 + xx + 16, 0), + __lsx_vld(src1 + xx, 0), + vperm), 0); + __m128i v11 = __lsx_vsllwil_hu_bu(__lsx_vshuf_b(__lsx_vld(src1 + xx + 17, 0), + __lsx_vld(src1 + xx + 1, 0), + vperm), 0); + __m128i v20 = __lsx_vsllwil_hu_bu(__lsx_vshuf_b(__lsx_vld(src2 + xx + 16, 0), + __lsx_vld(src2 + xx, 0), + vperm), 0); + __m128i v21 = __lsx_vsllwil_hu_bu(__lsx_vshuf_b(__lsx_vld(src2 + xx + 17, 0), + __lsx_vld(src2 + xx + 1, 0), + vperm), 0); + + __m128i w0 = __lsx_vsub_h(v127, valpha); + __lsx_vst(__lsx_vadd_h(__lsx_vmul_h(v10, w0), __lsx_vmul_h(v11, valpha)), + dst1 + i, 0); + __lsx_vst(__lsx_vadd_h(__lsx_vmul_h(v20, w0), __lsx_vmul_h(v21, valpha)), + dst2 + i, 0); + } + } + + for (; i < dstWidth; i++) { + unsigned int xx = xpos >> 16; + unsigned int xalpha = (xpos & 0xFFFF) >> 9; + dst1[i] = (src1[xx] * (xalpha ^ 127) + src1[xx + 1] * xalpha); + dst2[i] = (src2[xx] * (xalpha ^ 127) + src2[xx + 1] * xalpha); + xpos += xInc; + } + for (i = dstWidth - 1; (i * (int64_t)xInc) >> 16 >= srcW - 1; i--) { + dst1[i] = src1[srcW - 1] * 128; + dst2[i] = src2[srcW - 1] * 128; + } +} diff --git a/libswscale/loongarch/swscale_init_loongarch.c b/libswscale/loongarch/swscale_init_loongarch.c index 0c937b047f..ced73f2b0f 100644 --- a/libswscale/loongarch/swscale_init_loongarch.c +++ b/libswscale/loongarch/swscale_init_loongarch.c @@ -71,6 +71,12 @@ av_cold void ff_sws_init_swscale_loongarch(SwsInternal *c) c->hyScale = c->hcScale = c->dstBpc > 14 ? ff_hscale_16_to_19_lsx : ff_hscale_16_to_15_lsx; } + if (c->srcBpc == 8 && c->dstBpc <= 14 && + c->opts.flags & SWS_FAST_BILINEAR && + c->lumXInc <= (1 << 18) && c->chrXInc <= (1 << 18)) { + c->hyscale_fast = ff_hyscale_fast_lsx; + c->hcscale_fast = ff_hcscale_fast_lsx; + } } #if HAVE_LASX if (have_lasx(cpu_flags)) { diff --git a/libswscale/loongarch/swscale_loongarch.h b/libswscale/loongarch/swscale_loongarch.h index a8297b6972..b2875ba70f 100644 --- a/libswscale/loongarch/swscale_loongarch.h +++ b/libswscale/loongarch/swscale_loongarch.h @@ -50,6 +50,13 @@ void ff_hscale_16_to_19_sub_lsx(SwsInternal *c, int16_t *_dst, int dstW, const uint8_t *_src, const int16_t *filter, const int32_t *filterPos, int filterSize, int sh); +void ff_hyscale_fast_lsx(SwsInternal *c, int16_t *dst, int dstWidth, + const uint8_t *src, int srcW, int xInc); + +void ff_hcscale_fast_lsx(SwsInternal *c, int16_t *dst1, int16_t *dst2, + int dstWidth, const uint8_t *src1, + const uint8_t *src2, int srcW, int xInc); + void lumRangeFromJpeg_lsx(int16_t *dst, int width, uint32_t coeff, int64_t offset); void chrRangeFromJpeg_lsx(int16_t *dstU, int16_t *dstV, int width, uint32_t coeff, int64_t offset); void lumRangeToJpeg_lsx(int16_t *dst, int width, uint32_t coeff, int64_t offset); diff --git a/tests/checkasm/sw_scale.c b/tests/checkasm/sw_scale.c index b06ac23392..68d8f8d735 100644 --- a/tests/checkasm/sw_scale.c +++ b/tests/checkasm/sw_scale.c @@ -492,10 +492,130 @@ static void check_hscale(void) sws_freeContext(sws); } +/* Fast-bilinear horizontal scaling (SWS_FAST_BILINEAR, c->hyscale_fast / + * c->hcscale_fast): the optimized implementation must be bit-exact with + * the C reference. Catches e.g. swapped gather-position/weight tables in + * the LoongArch LSX implementation, which no FATE frame test covers. + * + * Only enabled on LoongArch: other archs' fast-bilinear implementations + * are not guaranteed bit-exact with the C reference (e.g. x86 MMXEXT uses + * complementary weights), so a cross-arch comparison would fail there. */ +#if ARCH_LOONGARCH64 +static void check_hyscale_fast(void) +{ +#define HSCALE_FAST_SRC_SIZE 4096 + static const int dstW_list[] = { 8, 16, 63, 100, 255, 256, 300, 511, 512, 1024 }; + static const int xInc_list[] = { 32768, 65536, 83886, 98304, 131072, 200000, 262143, 262144 }; + LOCAL_ALIGNED_32(uint8_t, src, [HSCALE_FAST_SRC_SIZE + 32]); + LOCAL_ALIGNED_32(int16_t, dst0, [2048]); + LOCAL_ALIGNED_32(int16_t, dst1, [2048]); + SwsContext *sws; + SwsInternal *c; + int i, j; + + declare_func(void, SwsInternal *c, int16_t *dst, int dstWidth, + const uint8_t *src, int srcW, int xInc); + + sws = sws_alloc_context(); + if (!sws || sws_init_context(sws, NULL, NULL) < 0) + fail(); + + c = sws_internal(sws); + c->srcBpc = 8; + c->dstBpc = 8; + c->opts.flags = SWS_FAST_BILINEAR; + c->lumXInc = c->chrXInc = 83886; + ff_sws_init_scale(c); + + if (check_func(c->hyscale_fast, "hyscale_fast")) { + for (i = 0; i < FF_ARRAY_ELEMS(dstW_list); i++) { + for (j = 0; j < FF_ARRAY_ELEMS(xInc_list); j++) { + int dstW = dstW_list[i]; + int xInc = xInc_list[j]; + int srcW = FFMIN(HSCALE_FAST_SRC_SIZE, + (dstW * xInc) >> 16); + + randomize_buffers(src, srcW + 16); + memset(dst0, 0, dstW * sizeof(dst0[0])); + memset(dst1, 0, dstW * sizeof(dst1[0])); + + call_ref(NULL, dst0, dstW, src, srcW, xInc); + call_new(NULL, dst1, dstW, src, srcW, xInc); + if (memcmp(dst0, dst1, dstW * sizeof(dst0[0]))) + fail(); + } + } + bench_new(NULL, dst1, 300, src, 512, 83886); + } + sws_freeContext(sws); +} + +static void check_hcscale_fast(void) +{ +#define HCSCALE_FAST_SRC_SIZE 4096 + static const int dstW_list[] = { 8, 16, 63, 100, 255, 256, 300, 511, 512, 1024 }; + static const int xInc_list[] = { 32768, 65536, 83886, 98304, 131072, 200000, 262143, 262144 }; + LOCAL_ALIGNED_32(uint8_t, src1, [HCSCALE_FAST_SRC_SIZE + 32]); + LOCAL_ALIGNED_32(uint8_t, src2, [HCSCALE_FAST_SRC_SIZE + 32]); + LOCAL_ALIGNED_32(int16_t, dst0, [2048]); + LOCAL_ALIGNED_32(int16_t, dst1, [2048]); + LOCAL_ALIGNED_32(int16_t, dst2, [2048]); + LOCAL_ALIGNED_32(int16_t, dst3, [2048]); + SwsContext *sws; + SwsInternal *c; + int i, j; + + declare_func(void, SwsInternal *c, int16_t *dst1, int16_t *dst2, int dstWidth, + const uint8_t *src1, const uint8_t *src2, int srcW, int xInc); + + sws = sws_alloc_context(); + if (!sws || sws_init_context(sws, NULL, NULL) < 0) + fail(); + + c = sws_internal(sws); + c->srcBpc = 8; + c->dstBpc = 8; + c->opts.flags = SWS_FAST_BILINEAR; + c->lumXInc = c->chrXInc = 83886; + ff_sws_init_scale(c); + + if (check_func(c->hcscale_fast, "hcscale_fast")) { + for (i = 0; i < FF_ARRAY_ELEMS(dstW_list); i++) { + for (j = 0; j < FF_ARRAY_ELEMS(xInc_list); j++) { + int dstW = dstW_list[i]; + int xInc = xInc_list[j]; + int srcW = FFMIN(HCSCALE_FAST_SRC_SIZE, + (dstW * xInc) >> 16); + + randomize_buffers(src1, srcW + 16); + randomize_buffers(src2, srcW + 16); + memset(dst0, 0, dstW * sizeof(dst0[0])); + memset(dst1, 0, dstW * sizeof(dst1[0])); + memset(dst2, 0, dstW * sizeof(dst2[0])); + memset(dst3, 0, dstW * sizeof(dst3[0])); + + call_ref(NULL, dst0, dst2, dstW, src1, src2, srcW, xInc); + call_new(NULL, dst1, dst3, dstW, src1, src2, srcW, xInc); + if (memcmp(dst0, dst1, dstW * sizeof(dst0[0])) || + memcmp(dst2, dst3, dstW * sizeof(dst2[0]))) + fail(); + } + } + bench_new(NULL, dst1, dst2, 300, src1, src2, 512, 83886); + } + sws_freeContext(sws); +} +#endif /* ARCH_LOONGARCH64 */ + void checkasm_check_sw_scale(void) { check_hscale(); report("hscale"); +#if ARCH_LOONGARCH64 + check_hyscale_fast(); + check_hcscale_fast(); + report("hscale_fast"); +#endif check_yuv2yuv1(0); check_yuv2yuv1(1); report("yuv2yuv1"); -- To stop receiving notification emails like this one, please contact [email protected]. _______________________________________________ ffmpeg-cvslog mailing list -- [email protected] To unsubscribe send an email to [email protected]
