This is an automated email from the git hooks/post-receive script.

Git pushed a commit to branch master
in repository ffmpeg.

commit edcdd87766b42ca514ecf7c728ef92ec62ec3202
Author:     Zuxy Meng <[email protected]>
AuthorDate: Fri Sep 4 21:31:57 2026 -0700
Commit:     Zuxy Meng <[email protected]>
CommitDate: Thu Oct 1 17:49:31 2026 -0700

    avcodec/riscv/h264dsp: Fix edge cases for bi-weight on RISCV
    
    The biweight sum (src*weights + dst*weightd + offset) must saturate to
    S16 after the dot product is complete, like the x86 SSE2 code does
    (paddsw dot product first, then paddsw offset). The RVV code added the
    offset first with a wrapping accumulate, so large-magnitude sums wrapped
    instead of saturating.
    
    Use vwmulsu.vv followed by vwmaccsu.vx and vsadd.vx to handle saturation
    in the correct order. This fixes bi-weight checkasm test for RISCV
    
    Signed-off-by: Zuxy Meng <[email protected]>
---
 libavcodec/riscv/h264dsp_rvv.S | 15 +++++++++------
 tests/checkasm/h264dsp.c       |  2 +-
 2 files changed, 10 insertions(+), 7 deletions(-)

diff --git a/libavcodec/riscv/h264dsp_rvv.S b/libavcodec/riscv/h264dsp_rvv.S
index 60015a7020..41bc72090c 100644
--- a/libavcodec/riscv/h264dsp_rvv.S
+++ b/libavcodec/riscv/h264dsp_rvv.S
@@ -64,17 +64,19 @@ func ff_h264_biweight_pixels_simple_8_rvv, zve32x
         ori     a7, a7, 1
         sll     a7, a7, a4
         addi    a4, a4, 1
+        vsetvli t0, zero, e8, m1, ta, ma
+        vmv.v.x v4, a5                      # splat weights (hoisted)
 1:
         vsetvli zero, t6, e16, m2, ta, ma
         vle8.v  v8, (a0)
         addi    a3, a3, -1
         vle8.v  v12, (a1)
         add     a1, a1, a2
-        vmv.v.x v16, a7
         vsetvli     zero, zero, e8, m1, ta, ma
-        vwmaccsu.vx v16, a5, v8
+        vwmulsu.vv  v16, v4, v8
         vwmaccsu.vx v16, a6, v12
         vsetvli     zero, zero, e16, m2, ta, ma
+        vsadd.vx    v16, v16, a7
         vmax.vx v16, v16, zero
         vsetvli zero, zero, e8, m1, ta, ma
         vnclipu.wx  v8, v16, a4
@@ -126,19 +128,20 @@ func ff_h264_biweight_pixels\w\()_\depth\()_rvv, zve64x
         ori     a7, a7, 1
         sll     a7, a7, a4
         addi    a4, a4, 1
+        vsetvli t0, zero, e8, m2, ta, ma
+        vmv.v.x v4, a5                      # splat weights (hoisted)
 1:
         vsetvli     t1, a3, e\b, m2, ta, ma
         vlse\b\().v v8, (a0), a2
         sub     a3, a3, t1
         vlse\b\().v v12, (a1), a2
         mul     t2, t1, a2
-        vsetvli     t0, zero, e16, m4, ta, ma
-        vmv.v.x     v16, a7
-        vsetvli     zero, zero, e8, m2, ta, ma
-        vwmaccsu.vx v16, a5, v8
+        vsetvli     t0, zero, e8, m2, ta, ma
+        vwmulsu.vv  v16, v4, v8
         add     a1, a1, t2
         vwmaccsu.vx v16, a6, v12
         vsetvli     zero, zero, e16, m4, ta, ma
+        vsadd.vx    v16, v16, a7
         vmax.vx     v16, v16, zero
         vsetvli     zero, zero, e8, m2, ta, ma
         vnclipu.wx  v8, v16, a4
diff --git a/tests/checkasm/h264dsp.c b/tests/checkasm/h264dsp.c
index 06d7cc325c..c410461f51 100644
--- a/tests/checkasm/h264dsp.c
+++ b/tests/checkasm/h264dsp.c
@@ -545,7 +545,7 @@ static void check_weight(void)
 }
 
 // only archs that can pass test
-#define H264_CHECK_BIWEIGHT (ARCH_X86 || ARCH_PPC || ARCH_MIPS || ARCH_ARM || 
ARCH_AARCH64)
+#define H264_CHECK_BIWEIGHT (ARCH_X86 || ARCH_PPC || ARCH_MIPS || ARCH_ARM || 
ARCH_AARCH64 || ARCH_RISCV)
 
 #if H264_CHECK_BIWEIGHT
 static void check_biweight(void)

-- 
To stop receiving notification emails like this one, please contact
[email protected].
_______________________________________________
ffmpeg-cvslog mailing list -- [email protected]
To unsubscribe send an email to [email protected]

Reply via email to