This is an automated email from the git hooks/post-receive script. Git pushed a commit to branch master in repository ffmpeg.
commit f68e1afc1b7cf4275d09f1a9026ff80228cf99a8 Author: Zuxy Meng <[email protected]> AuthorDate: Fri Sep 4 21:35:36 2026 -0700 Commit: Zuxy Meng <[email protected]> CommitDate: Thu Oct 1 17:49:31 2026 -0700 avcodec/loongarch/h264dsp: Fix edge cases for bi-weight on loongarch S16 saturating sum must be computed dot-product-first. The LSX/LASX code accumulated the offset into the dot product with wrapping vmaddwev/wod, so large-magnitude sums wrapped instead of saturating. Accumulate the dot product from zero (wrapping is exact: the dot product cannot overflow S16 for 8-bit with log2_denom < 7), then add the offset with a saturating vsadd. Also tweaked the checkasm test itself to cover only valid width and height combo. This fixes bi-weight checkasm test for all currently available asm implementations. Signed-off-by: Zuxy Meng <[email protected]> --- libavcodec/loongarch/h264dsp.S | 27 ++++++++++++++++++--------- tests/checkasm/h264dsp.c | 13 ++++--------- 2 files changed, 22 insertions(+), 18 deletions(-) diff --git a/libavcodec/loongarch/h264dsp.S b/libavcodec/loongarch/h264dsp.S index 750fe49143..e7d4892cf3 100644 --- a/libavcodec/loongarch/h264dsp.S +++ b/libavcodec/loongarch/h264dsp.S @@ -943,10 +943,10 @@ endfunc .macro biweight_calc _in0, _in1, _in2, _in3, _reg0, _reg1, _reg2,\ _out0, _out1, _out2, _out3 - vmov \_out0, \_reg0 - vmov \_out1, \_reg0 - vmov \_out2, \_reg0 - vmov \_out3, \_reg0 + vreplgr2vr.h \_out0, zero + vreplgr2vr.h \_out1, zero + vreplgr2vr.h \_out2, zero + vreplgr2vr.h \_out3, zero vmaddwev.h.bu.b \_out0, \_in0, \_reg1 vmaddwev.h.bu.b \_out1, \_in1, \_reg1 vmaddwev.h.bu.b \_out2, \_in2, \_reg1 @@ -955,6 +955,10 @@ endfunc vmaddwod.h.bu.b \_out1, \_in1, \_reg1 vmaddwod.h.bu.b \_out2, \_in2, \_reg1 vmaddwod.h.bu.b \_out3, \_in3, \_reg1 + vsadd.h \_out0, \_out0, \_reg0 + vsadd.h \_out1, \_out1, \_reg0 + vsadd.h \_out2, \_out2, \_reg0 + vsadd.h \_out3, \_out3, \_reg0 vssran.bu.h \_out0, \_out0, \_reg2 vssran.bu.h \_out1, \_out1, \_reg2 @@ -1133,9 +1137,10 @@ biweight_func 16 endfunc .macro biweight_calc_4 _in0, _out0 - vmov \_out0, vr8 + vreplgr2vr.h \_out0, zero vmaddwev.h.bu.b \_out0, \_in0, vr20 vmaddwod.h.bu.b \_out0, \_in0, vr20 + vsadd.h \_out0, \_out0, vr8 vssran.bu.h \_out0, \_out0, vr9 .endm @@ -1188,12 +1193,14 @@ biweight_func 4 vilvl.b vr0, vr14, vr4 vilvl.b vr10, vr15, vr5 - vmov vr1, vr8 - vmov vr11, vr8 + vreplgr2vr.h vr1, zero + vreplgr2vr.h vr11, zero vmaddwev.h.bu.b vr1, vr0, vr20 vmaddwev.h.bu.b vr11, vr10, vr20 vmaddwod.h.bu.b vr1, vr0, vr20 vmaddwod.h.bu.b vr11, vr10, vr20 + vsadd.h vr1, vr1, vr8 + vsadd.h vr11, vr11, vr8 vssran.bu.h vr0, vr1, vr9 //vec0 vssran.bu.h vr10, vr11, vr9 //vec0 @@ -1225,12 +1232,14 @@ function ff_biweight_h264_pixels\w\()_8_lasx .endm .macro biweight_calc_lasx _in0, _in1, _reg0, _reg1, _reg2, _out0, _out1 - xmov \_out0, \_reg0 - xmov \_out1, \_reg0 + xvreplgr2vr.h \_out0, zero + xvreplgr2vr.h \_out1, zero xvmaddwev.h.bu.b \_out0, \_in0, \_reg1 xvmaddwev.h.bu.b \_out1, \_in1, \_reg1 xvmaddwod.h.bu.b \_out0, \_in0, \_reg1 xvmaddwod.h.bu.b \_out1, \_in1, \_reg1 + xvsadd.h \_out0, \_out0, \_reg0 + xvsadd.h \_out1, \_out1, \_reg0 xvssran.bu.h \_out0, \_out0, \_reg2 xvssran.bu.h \_out1, \_out1, \_reg2 diff --git a/tests/checkasm/h264dsp.c b/tests/checkasm/h264dsp.c index c410461f51..3bd0dec2c3 100644 --- a/tests/checkasm/h264dsp.c +++ b/tests/checkasm/h264dsp.c @@ -517,7 +517,8 @@ static void check_weight(void) if (check_func(h.weight_pixels_tab[idx], "weight_%dx%d_%d", w, 16, bit_depth)) { - for (int hgt = 16; hgt >= 2; hgt >>= 1) { + for (int hgt = FFMIN(16, 2 * w); + hgt >= FFMAX(2, w / 2); hgt >>= 1) { for (int i = 0; i < 32; i++) { int stride = 32 * SIZEOF_PIXEL; int log2_denom = rnd() % 8; @@ -544,10 +545,6 @@ static void check_weight(void) } } -// only archs that can pass test -#define H264_CHECK_BIWEIGHT (ARCH_X86 || ARCH_PPC || ARCH_MIPS || ARCH_ARM || ARCH_AARCH64 || ARCH_RISCV) - -#if H264_CHECK_BIWEIGHT static void check_biweight(void) { LOCAL_ALIGNED_16(uint8_t, dst, [32 * 32 * 2]); @@ -568,7 +565,8 @@ static void check_biweight(void) if (check_func(h.biweight_pixels_tab[idx], "biweight_%dx%d_%d", w, 16, bit_depth)) { - for (int hgt = 16; hgt >= 2; hgt >>= 1) { + for (int hgt = FFMIN(16, 2 * w); + hgt >= FFMAX(2, w / 2); hgt >>= 1) { for (int i = 0; i < 32; i++) { int stride = 32 * SIZEOF_PIXEL; // Spec allows for 0 <= log2_denom <= 7 regardless @@ -614,7 +612,6 @@ static void check_biweight(void) } } } -#endif void checkasm_check_h264dsp(void) { @@ -632,8 +629,6 @@ void checkasm_check_h264dsp(void) check_weight(); report("weight"); -#if H264_CHECK_BIWEIGHT check_biweight(); report("biweight"); -#endif } -- To stop receiving notification emails like this one, please contact [email protected]. _______________________________________________ ffmpeg-cvslog mailing list -- [email protected] To unsubscribe send an email to [email protected]
