PR #24198 opened by zuxy
URL: https://code.ffmpeg.org/FFmpeg/FFmpeg/pulls/24198
Patch URL: https://code.ffmpeg.org/FFmpeg/FFmpeg/pulls/24198.patch

Use SSE2 for (put|avg)[48]; identical performance.

Signed-off-by: Zuxy Meng <[email protected]>



>From ed42e124358c71d51e400d8aff1cb527fac0fd41 Mon Sep 17 00:00:00 2001
From: Zuxy Meng <[email protected]>
Date: Sun, 16 Aug 2026 17:12:41 -0700
Subject: [PATCH] avcodec/x86/vp9mc.asm: Replace MMX with SSE2

Use SSE2 for (put|avg)[48]; identical performance.

Signed-off-by: Zuxy Meng <[email protected]>
---
 libavcodec/x86/vp9dsp_init.c       | 16 ++++++++--------
 libavcodec/x86/vp9dsp_init_16bpp.c |  8 ++++----
 libavcodec/x86/vp9mc.asm           | 23 +++++++++++++++--------
 tests/checkasm/vp9dsp.c            |  2 +-
 4 files changed, 28 insertions(+), 21 deletions(-)

diff --git a/libavcodec/x86/vp9dsp_init.c b/libavcodec/x86/vp9dsp_init.c
index e479fd25ee..7b2fd27203 100644
--- a/libavcodec/x86/vp9dsp_init.c
+++ b/libavcodec/x86/vp9dsp_init.c
@@ -26,13 +26,13 @@
 #include "libavcodec/vp9dsp.h"
 #include "libavcodec/x86/vp9dsp_init.h"
 
-decl_fpel_func(put,  4,   , mmx);
-decl_fpel_func(put,  8,   , mmx);
+decl_fpel_func(put,  4,   , sse2);
+decl_fpel_func(put,  8,   , sse2);
 decl_fpel_func(put, 16,   , sse);
 decl_fpel_func(put, 32,   , sse);
 decl_fpel_func(put, 64,   , sse);
-decl_fpel_func(avg,  4, _8, mmxext);
-decl_fpel_func(avg,  8, _8, mmxext);
+decl_fpel_func(avg,  4, _8, sse2);
+decl_fpel_func(avg,  8, _8, sse2);
 decl_fpel_func(avg, 16, _8, sse2);
 decl_fpel_func(avg, 32, _8, sse2);
 decl_fpel_func(avg, 64, _8, sse2);
@@ -265,8 +265,6 @@ av_cold void ff_vp9dsp_init_x86(VP9DSPContext *dsp, int 
bpp, int bitexact)
 } while (0)
 
     if (EXTERNAL_MMX(cpu_flags)) {
-        init_fpel_func(4, 0,  4, put, , mmx);
-        init_fpel_func(3, 0,  8, put, , mmx);
         if (!bitexact) {
             dsp->itxfm_add[4 /* lossless */][DCT_DCT] =
             dsp->itxfm_add[4 /* lossless */][ADST_DCT] =
@@ -281,8 +279,6 @@ av_cold void ff_vp9dsp_init_x86(VP9DSPContext *dsp, int 
bpp, int bitexact)
         dsp->loop_filter_8[0][1] = ff_vp9_loop_filter_v_4_8_mmxext;
         dsp->loop_filter_8[1][0] = ff_vp9_loop_filter_h_8_8_mmxext;
         dsp->loop_filter_8[1][1] = ff_vp9_loop_filter_v_8_8_mmxext;
-        init_fpel_func(4, 1,  4, avg, _8, mmxext);
-        init_fpel_func(3, 1,  8, avg, _8, mmxext);
         dsp->intra_pred[TX_4X4][HOR_DOWN_PRED] = ff_vp9_ipred_hd_4x4_mmxext;
         dsp->intra_pred[TX_4X4][VERT_LEFT_PRED] = ff_vp9_ipred_vl_4x4_mmxext;
     }
@@ -298,6 +294,10 @@ av_cold void ff_vp9dsp_init_x86(VP9DSPContext *dsp, int 
bpp, int bitexact)
     if (EXTERNAL_SSE2(cpu_flags)) {
         init_subpel3_8to64(0, put, 8, sse2);
         init_subpel3_8to64(1, avg, 8, sse2);
+        init_fpel_func(4, 0,  4, put, , sse2);
+        init_fpel_func(3, 0,  8, put, , sse2);
+        init_fpel_func(4, 1,  4, avg, _8, sse2);
+        init_fpel_func(3, 1,  8, avg, _8, sse2);
         init_fpel_func(2, 1, 16, avg,  _8, sse2);
         init_fpel_func(1, 1, 32, avg,  _8, sse2);
         init_fpel_func(0, 1, 64, avg,  _8, sse2);
diff --git a/libavcodec/x86/vp9dsp_init_16bpp.c 
b/libavcodec/x86/vp9dsp_init_16bpp.c
index 2d2f01ba5f..3748c316b0 100644
--- a/libavcodec/x86/vp9dsp_init_16bpp.c
+++ b/libavcodec/x86/vp9dsp_init_16bpp.c
@@ -26,8 +26,8 @@
 #include "libavcodec/vp9dsp.h"
 #include "libavcodec/x86/vp9dsp_init.h"
 
-decl_fpel_func(put,   8,    , mmx);
-decl_fpel_func(avg,   8, _16, mmxext);
+decl_fpel_func(put,   8,    , sse2);
+decl_fpel_func(avg,   8, _16, sse2);
 decl_fpel_func(put,  16,    , sse);
 decl_fpel_func(put,  32,    , sse);
 decl_fpel_func(put,  64,    , sse);
@@ -72,12 +72,10 @@ av_cold void ff_vp9dsp_init_16bpp_x86(VP9DSPContext *dsp)
     int cpu_flags = av_get_cpu_flags();
 
     if (EXTERNAL_MMX(cpu_flags)) {
-        init_fpel_func(4, 0,   8, put, , mmx);
         init_ipred_func(v, VERT, 4, 16, mmx);
     }
 
     if (EXTERNAL_MMXEXT(cpu_flags)) {
-        init_fpel_func(4, 1,   8, avg, _16, mmxext);
         init_ipred_func(h, HOR, 4, 16, mmxext);
         init_ipred_func(dc, DC, 4, 16, mmxext);
         init_ipred_func(dc_top,  TOP_DC,  4, 16, mmxext);
@@ -93,6 +91,8 @@ av_cold void ff_vp9dsp_init_16bpp_x86(VP9DSPContext *dsp)
     }
 
     if (EXTERNAL_SSE2(cpu_flags)) {
+        init_fpel_func(4, 0,   8, put, , sse2);
+        init_fpel_func(4, 1,   8, avg, _16, sse2);
         init_fpel_func(3, 1,  16, avg, _16, sse2);
         init_fpel_func(2, 1,  32, avg, _16, sse2);
         init_fpel_func(1, 1,  64, avg, _16, sse2);
diff --git a/libavcodec/x86/vp9mc.asm b/libavcodec/x86/vp9mc.asm
index 0e6aa627db..bd38197eed 100644
--- a/libavcodec/x86/vp9mc.asm
+++ b/libavcodec/x86/vp9mc.asm
@@ -722,6 +722,9 @@ filter_vx2_fn avg
 
 %macro fpel_fn 6-8 0, 4
 %if %2 == 4
+%define %%srcfn movd
+%define %%dstfn movd
+%elif mmsize/%2 == 2
 %define %%srcfn movh
 %define %%dstfn movh
 %else
@@ -759,13 +762,19 @@ cglobal vp9_%1%2 %+ %%szsuf, 5, 5, %8, dst, dstride, src, 
sstride, h
 %endif
     lea       srcq, [srcq+sstrideq*%6]
 %ifidn %1, avg
-    %%pavg      m0, [dstq]
-    %%pavg      m1, [dstq+d%3]
-    %%pavg      m2, [dstq+d%4]
-%if %2 == 4
+%if %2 < mmsize
+    %%srcfn     m4, [dstq]
+    %%pavg      m0, m4
+    %%srcfn     m4, [dstq+d%3]
+    %%pavg      m1, m4
+    %%srcfn     m4, [dstq+d%4]
+    %%pavg      m2, m4
     %%srcfn     m4, [dstq+d%5]
     %%pavg      m3, m4
 %else
+    %%pavg      m0, [dstq]
+    %%pavg      m1, [dstq+d%3]
+    %%pavg      m2, [dstq+d%4]
     %%pavg      m3, [dstq+d%5]
 %endif
 %if %2/mmsize == 8
@@ -795,10 +804,9 @@ cglobal vp9_%1%2 %+ %%szsuf, 5, 5, %8, dst, dstride, src, 
sstride, h
 %define s16 16
 %define d32 32
 %define s32 32
-INIT_MMX mmx
+INIT_XMM sse2
 fpel_fn put, 4,  strideq, strideq*2, stride3q, 4
 fpel_fn put, 8,  strideq, strideq*2, stride3q, 4
-INIT_MMX mmxext
 fpel_fn avg, 4,  strideq, strideq*2, stride3q, 4, 8
 fpel_fn avg, 8,  strideq, strideq*2, stride3q, 4, 8
 INIT_XMM sse
@@ -819,9 +827,8 @@ INIT_YMM avx2
 fpel_fn avg, 32, strideq, strideq*2, stride3q, 4, 8
 fpel_fn avg, 64, mmsize,  strideq,   strideq+mmsize, 2, 8
 %endif
-INIT_MMX mmxext
-fpel_fn avg,  8,  strideq, strideq*2, stride3q, 4, 16
 INIT_XMM sse2
+fpel_fn avg,  8,  strideq, strideq*2, stride3q, 4, 16
 fpel_fn avg,  16, strideq, strideq*2, stride3q, 4, 16
 fpel_fn avg,  32, mmsize,  strideq,   strideq+mmsize, 2, 16
 fpel_fn avg,  64, mmsize,  mmsize*2,  mmsize*3, 1, 16
diff --git a/tests/checkasm/vp9dsp.c b/tests/checkasm/vp9dsp.c
index 2a3374541f..993e0715aa 100644
--- a/tests/checkasm/vp9dsp.c
+++ b/tests/checkasm/vp9dsp.c
@@ -563,7 +563,7 @@ static void check_mc(void)
     LOCAL_ALIGNED_64(uint8_t, dst1, [64 * 64 * 2]);
     VP9DSPContext dsp;
     int op, hsize, bit_depth, filter, dx, dy;
-    declare_func_emms(AV_CPU_FLAG_MMX | AV_CPU_FLAG_MMXEXT, void, uint8_t 
*dst, ptrdiff_t dst_stride,
+    declare_func(void, uint8_t *dst, ptrdiff_t dst_stride,
                       const uint8_t *ref, ptrdiff_t ref_stride,
                  int h, int mx, int my);
     static const char *const filter_names[4] = {
-- 
2.52.0

_______________________________________________
ffmpeg-devel mailing list -- [email protected]
To unsubscribe send an email to [email protected]

Reply via email to