This is an automated email from the git hooks/post-receive script.
Git pushed a commit to branch master
in repository ffmpeg.
The following commit(s) were added to refs/heads/master by this push:
new 2a168907d3 avcodec/x86/vp9mc.asm: Replace MMX with SSE2
2a168907d3 is described below
commit 2a168907d36e7fbbb782b06d527c20ddefc1fccd
Author: Zuxy Meng <[email protected]>
AuthorDate: Sun Aug 16 17:12:41 2026 -0700
Commit: Zuxy Meng <[email protected]>
CommitDate: Wed Aug 19 23:28:47 2026 -0700
avcodec/x86/vp9mc.asm: Replace MMX with SSE2
Use SSE2 for (put|avg)[48]; identical performance.
Signed-off-by: Zuxy Meng <[email protected]>
---
libavcodec/x86/vp9dsp_init.c | 16 ++++++++--------
libavcodec/x86/vp9dsp_init_16bpp.c | 8 ++++----
libavcodec/x86/vp9mc.asm | 23 +++++++++++++++--------
tests/checkasm/vp9dsp.c | 2 +-
4 files changed, 28 insertions(+), 21 deletions(-)
diff --git a/libavcodec/x86/vp9dsp_init.c b/libavcodec/x86/vp9dsp_init.c
index e479fd25ee..7b2fd27203 100644
--- a/libavcodec/x86/vp9dsp_init.c
+++ b/libavcodec/x86/vp9dsp_init.c
@@ -26,13 +26,13 @@
#include "libavcodec/vp9dsp.h"
#include "libavcodec/x86/vp9dsp_init.h"
-decl_fpel_func(put, 4, , mmx);
-decl_fpel_func(put, 8, , mmx);
+decl_fpel_func(put, 4, , sse2);
+decl_fpel_func(put, 8, , sse2);
decl_fpel_func(put, 16, , sse);
decl_fpel_func(put, 32, , sse);
decl_fpel_func(put, 64, , sse);
-decl_fpel_func(avg, 4, _8, mmxext);
-decl_fpel_func(avg, 8, _8, mmxext);
+decl_fpel_func(avg, 4, _8, sse2);
+decl_fpel_func(avg, 8, _8, sse2);
decl_fpel_func(avg, 16, _8, sse2);
decl_fpel_func(avg, 32, _8, sse2);
decl_fpel_func(avg, 64, _8, sse2);
@@ -265,8 +265,6 @@ av_cold void ff_vp9dsp_init_x86(VP9DSPContext *dsp, int
bpp, int bitexact)
} while (0)
if (EXTERNAL_MMX(cpu_flags)) {
- init_fpel_func(4, 0, 4, put, , mmx);
- init_fpel_func(3, 0, 8, put, , mmx);
if (!bitexact) {
dsp->itxfm_add[4 /* lossless */][DCT_DCT] =
dsp->itxfm_add[4 /* lossless */][ADST_DCT] =
@@ -281,8 +279,6 @@ av_cold void ff_vp9dsp_init_x86(VP9DSPContext *dsp, int
bpp, int bitexact)
dsp->loop_filter_8[0][1] = ff_vp9_loop_filter_v_4_8_mmxext;
dsp->loop_filter_8[1][0] = ff_vp9_loop_filter_h_8_8_mmxext;
dsp->loop_filter_8[1][1] = ff_vp9_loop_filter_v_8_8_mmxext;
- init_fpel_func(4, 1, 4, avg, _8, mmxext);
- init_fpel_func(3, 1, 8, avg, _8, mmxext);
dsp->intra_pred[TX_4X4][HOR_DOWN_PRED] = ff_vp9_ipred_hd_4x4_mmxext;
dsp->intra_pred[TX_4X4][VERT_LEFT_PRED] = ff_vp9_ipred_vl_4x4_mmxext;
}
@@ -298,6 +294,10 @@ av_cold void ff_vp9dsp_init_x86(VP9DSPContext *dsp, int
bpp, int bitexact)
if (EXTERNAL_SSE2(cpu_flags)) {
init_subpel3_8to64(0, put, 8, sse2);
init_subpel3_8to64(1, avg, 8, sse2);
+ init_fpel_func(4, 0, 4, put, , sse2);
+ init_fpel_func(3, 0, 8, put, , sse2);
+ init_fpel_func(4, 1, 4, avg, _8, sse2);
+ init_fpel_func(3, 1, 8, avg, _8, sse2);
init_fpel_func(2, 1, 16, avg, _8, sse2);
init_fpel_func(1, 1, 32, avg, _8, sse2);
init_fpel_func(0, 1, 64, avg, _8, sse2);
diff --git a/libavcodec/x86/vp9dsp_init_16bpp.c
b/libavcodec/x86/vp9dsp_init_16bpp.c
index 2d2f01ba5f..3748c316b0 100644
--- a/libavcodec/x86/vp9dsp_init_16bpp.c
+++ b/libavcodec/x86/vp9dsp_init_16bpp.c
@@ -26,8 +26,8 @@
#include "libavcodec/vp9dsp.h"
#include "libavcodec/x86/vp9dsp_init.h"
-decl_fpel_func(put, 8, , mmx);
-decl_fpel_func(avg, 8, _16, mmxext);
+decl_fpel_func(put, 8, , sse2);
+decl_fpel_func(avg, 8, _16, sse2);
decl_fpel_func(put, 16, , sse);
decl_fpel_func(put, 32, , sse);
decl_fpel_func(put, 64, , sse);
@@ -72,12 +72,10 @@ av_cold void ff_vp9dsp_init_16bpp_x86(VP9DSPContext *dsp)
int cpu_flags = av_get_cpu_flags();
if (EXTERNAL_MMX(cpu_flags)) {
- init_fpel_func(4, 0, 8, put, , mmx);
init_ipred_func(v, VERT, 4, 16, mmx);
}
if (EXTERNAL_MMXEXT(cpu_flags)) {
- init_fpel_func(4, 1, 8, avg, _16, mmxext);
init_ipred_func(h, HOR, 4, 16, mmxext);
init_ipred_func(dc, DC, 4, 16, mmxext);
init_ipred_func(dc_top, TOP_DC, 4, 16, mmxext);
@@ -93,6 +91,8 @@ av_cold void ff_vp9dsp_init_16bpp_x86(VP9DSPContext *dsp)
}
if (EXTERNAL_SSE2(cpu_flags)) {
+ init_fpel_func(4, 0, 8, put, , sse2);
+ init_fpel_func(4, 1, 8, avg, _16, sse2);
init_fpel_func(3, 1, 16, avg, _16, sse2);
init_fpel_func(2, 1, 32, avg, _16, sse2);
init_fpel_func(1, 1, 64, avg, _16, sse2);
diff --git a/libavcodec/x86/vp9mc.asm b/libavcodec/x86/vp9mc.asm
index 0e6aa627db..bd38197eed 100644
--- a/libavcodec/x86/vp9mc.asm
+++ b/libavcodec/x86/vp9mc.asm
@@ -722,6 +722,9 @@ filter_vx2_fn avg
%macro fpel_fn 6-8 0, 4
%if %2 == 4
+%define %%srcfn movd
+%define %%dstfn movd
+%elif mmsize/%2 == 2
%define %%srcfn movh
%define %%dstfn movh
%else
@@ -759,13 +762,19 @@ cglobal vp9_%1%2 %+ %%szsuf, 5, 5, %8, dst, dstride, src,
sstride, h
%endif
lea srcq, [srcq+sstrideq*%6]
%ifidn %1, avg
- %%pavg m0, [dstq]
- %%pavg m1, [dstq+d%3]
- %%pavg m2, [dstq+d%4]
-%if %2 == 4
+%if %2 < mmsize
+ %%srcfn m4, [dstq]
+ %%pavg m0, m4
+ %%srcfn m4, [dstq+d%3]
+ %%pavg m1, m4
+ %%srcfn m4, [dstq+d%4]
+ %%pavg m2, m4
%%srcfn m4, [dstq+d%5]
%%pavg m3, m4
%else
+ %%pavg m0, [dstq]
+ %%pavg m1, [dstq+d%3]
+ %%pavg m2, [dstq+d%4]
%%pavg m3, [dstq+d%5]
%endif
%if %2/mmsize == 8
@@ -795,10 +804,9 @@ cglobal vp9_%1%2 %+ %%szsuf, 5, 5, %8, dst, dstride, src,
sstride, h
%define s16 16
%define d32 32
%define s32 32
-INIT_MMX mmx
+INIT_XMM sse2
fpel_fn put, 4, strideq, strideq*2, stride3q, 4
fpel_fn put, 8, strideq, strideq*2, stride3q, 4
-INIT_MMX mmxext
fpel_fn avg, 4, strideq, strideq*2, stride3q, 4, 8
fpel_fn avg, 8, strideq, strideq*2, stride3q, 4, 8
INIT_XMM sse
@@ -819,9 +827,8 @@ INIT_YMM avx2
fpel_fn avg, 32, strideq, strideq*2, stride3q, 4, 8
fpel_fn avg, 64, mmsize, strideq, strideq+mmsize, 2, 8
%endif
-INIT_MMX mmxext
-fpel_fn avg, 8, strideq, strideq*2, stride3q, 4, 16
INIT_XMM sse2
+fpel_fn avg, 8, strideq, strideq*2, stride3q, 4, 16
fpel_fn avg, 16, strideq, strideq*2, stride3q, 4, 16
fpel_fn avg, 32, mmsize, strideq, strideq+mmsize, 2, 16
fpel_fn avg, 64, mmsize, mmsize*2, mmsize*3, 1, 16
diff --git a/tests/checkasm/vp9dsp.c b/tests/checkasm/vp9dsp.c
index 2a3374541f..993e0715aa 100644
--- a/tests/checkasm/vp9dsp.c
+++ b/tests/checkasm/vp9dsp.c
@@ -563,7 +563,7 @@ static void check_mc(void)
LOCAL_ALIGNED_64(uint8_t, dst1, [64 * 64 * 2]);
VP9DSPContext dsp;
int op, hsize, bit_depth, filter, dx, dy;
- declare_func_emms(AV_CPU_FLAG_MMX | AV_CPU_FLAG_MMXEXT, void, uint8_t
*dst, ptrdiff_t dst_stride,
+ declare_func(void, uint8_t *dst, ptrdiff_t dst_stride,
const uint8_t *ref, ptrdiff_t ref_stride,
int h, int mx, int my);
static const char *const filter_names[4] = {
--
To stop receiving notification emails like this one, please contact
[email protected].
_______________________________________________
ffmpeg-cvslog mailing list -- [email protected]
To unsubscribe send an email to [email protected]