This is an automated email from the git hooks/post-receive script. Git pushed a commit to branch master in repository ffmpeg.
commit edcdd87766b42ca514ecf7c728ef92ec62ec3202 Author: Zuxy Meng <[email protected]> AuthorDate: Fri Sep 4 21:31:57 2026 -0700 Commit: Zuxy Meng <[email protected]> CommitDate: Thu Oct 1 17:49:31 2026 -0700 avcodec/riscv/h264dsp: Fix edge cases for bi-weight on RISCV The biweight sum (src*weights + dst*weightd + offset) must saturate to S16 after the dot product is complete, like the x86 SSE2 code does (paddsw dot product first, then paddsw offset). The RVV code added the offset first with a wrapping accumulate, so large-magnitude sums wrapped instead of saturating. Use vwmulsu.vv followed by vwmaccsu.vx and vsadd.vx to handle saturation in the correct order. This fixes bi-weight checkasm test for RISCV Signed-off-by: Zuxy Meng <[email protected]> --- libavcodec/riscv/h264dsp_rvv.S | 15 +++++++++------ tests/checkasm/h264dsp.c | 2 +- 2 files changed, 10 insertions(+), 7 deletions(-) diff --git a/libavcodec/riscv/h264dsp_rvv.S b/libavcodec/riscv/h264dsp_rvv.S index 60015a7020..41bc72090c 100644 --- a/libavcodec/riscv/h264dsp_rvv.S +++ b/libavcodec/riscv/h264dsp_rvv.S @@ -64,17 +64,19 @@ func ff_h264_biweight_pixels_simple_8_rvv, zve32x ori a7, a7, 1 sll a7, a7, a4 addi a4, a4, 1 + vsetvli t0, zero, e8, m1, ta, ma + vmv.v.x v4, a5 # splat weights (hoisted) 1: vsetvli zero, t6, e16, m2, ta, ma vle8.v v8, (a0) addi a3, a3, -1 vle8.v v12, (a1) add a1, a1, a2 - vmv.v.x v16, a7 vsetvli zero, zero, e8, m1, ta, ma - vwmaccsu.vx v16, a5, v8 + vwmulsu.vv v16, v4, v8 vwmaccsu.vx v16, a6, v12 vsetvli zero, zero, e16, m2, ta, ma + vsadd.vx v16, v16, a7 vmax.vx v16, v16, zero vsetvli zero, zero, e8, m1, ta, ma vnclipu.wx v8, v16, a4 @@ -126,19 +128,20 @@ func ff_h264_biweight_pixels\w\()_\depth\()_rvv, zve64x ori a7, a7, 1 sll a7, a7, a4 addi a4, a4, 1 + vsetvli t0, zero, e8, m2, ta, ma + vmv.v.x v4, a5 # splat weights (hoisted) 1: vsetvli t1, a3, e\b, m2, ta, ma vlse\b\().v v8, (a0), a2 sub a3, a3, t1 vlse\b\().v v12, (a1), a2 mul t2, t1, a2 - vsetvli t0, zero, e16, m4, ta, ma - vmv.v.x v16, a7 - vsetvli zero, zero, e8, m2, ta, ma - vwmaccsu.vx v16, a5, v8 + vsetvli t0, zero, e8, m2, ta, ma + vwmulsu.vv v16, v4, v8 add a1, a1, t2 vwmaccsu.vx v16, a6, v12 vsetvli zero, zero, e16, m4, ta, ma + vsadd.vx v16, v16, a7 vmax.vx v16, v16, zero vsetvli zero, zero, e8, m2, ta, ma vnclipu.wx v8, v16, a4 diff --git a/tests/checkasm/h264dsp.c b/tests/checkasm/h264dsp.c index 06d7cc325c..c410461f51 100644 --- a/tests/checkasm/h264dsp.c +++ b/tests/checkasm/h264dsp.c @@ -545,7 +545,7 @@ static void check_weight(void) } // only archs that can pass test -#define H264_CHECK_BIWEIGHT (ARCH_X86 || ARCH_PPC || ARCH_MIPS || ARCH_ARM || ARCH_AARCH64) +#define H264_CHECK_BIWEIGHT (ARCH_X86 || ARCH_PPC || ARCH_MIPS || ARCH_ARM || ARCH_AARCH64 || ARCH_RISCV) #if H264_CHECK_BIWEIGHT static void check_biweight(void) -- To stop receiving notification emails like this one, please contact [email protected]. _______________________________________________ ffmpeg-cvslog mailing list -- [email protected] To unsubscribe send an email to [email protected]
