PR #24052 opened by Marcos Ashton (MarcosAsh)
URL: https://code.ffmpeg.org/FFmpeg/FFmpeg/pulls/24052
Patch URL: https://code.ffmpeg.org/FFmpeg/FFmpeg/pulls/24052.patch

Adds an x86 SIMD implementation of HEVC planar intra prediction (previously 
C-only), for 8 bit, as part of #23022. x86 has no HEVC intra prediction at all 
today; aarch64 NEON already implements the planar, DC and angular groups at 8 
bit.

Everything in the planar expression except (size-1-x)*left[y] is linear in y, 
so for a fixed group of eight columns the remaining terms collapse into a 
single running accumulator whose increment left[size]-top[x] is loop invariant. 
Iterating columns outermost and rows innermost keeps that accumulator and its 
delta in registers, so a row costs one pmullw and two paddw with no reload of 
top[]. For 16x16 and 32x32 two column groups run together, which shares the 
left[y] broadcast and lets a row be written with a single 16 byte store. 
Accumulation is in words throughout since the largest value the expression can 
take is 31*255 + 32*255 + 31*255 + 32*255 + 32 = 32162. SSSE3 is the baseline 
rather than SSE2 because pshufb does both the byte broadcasts and the byte to 
word widening, which keeps the row loop short and avoids a dedicated zero 
register. 4x4 keeps a separate fully unrolled path that does two rows per 
register.

checkasm -t hevc_pred passes bit-exact. Benchmarks (Core Ultra 7 155H, medians 
of five runs):
```
hevc_pred_planar_4x4_8_c:        18.4
hevc_pred_planar_4x4_8_ssse3:     5.5
hevc_pred_planar_8x8_8_c:        83.1
hevc_pred_planar_8x8_8_ssse3:    16.3
hevc_pred_planar_16x16_8_c:     224.8
hevc_pred_planar_16x16_8_ssse3:  47.3
hevc_pred_planar_32x32_8_c:     629.5
hevc_pred_planar_32x32_8_ssse3: 196.5
```


>From 47d10c66b9c9b35fd5f2dd6b1289a0183488a0f2 Mon Sep 17 00:00:00 2001
From: Marcos Ashton Iglesias <[email protected]>
Date: Fri, 7 Aug 2026 20:42:41 +0100
Subject: [PATCH] avcodec/x86/hevc: add SSSE3 pred_planar

Implements planar intra prediction for 8 bit. x86 has no HEVC intra
prediction at all; aarch64 NEON already covers planar, DC and angular.

Every term except (size-1-x)*left[y] is linear in y, so for a group of
eight columns they collapse into one running accumulator whose increment
left[size]-top[x] is loop invariant. Running columns outermost keeps that
accumulator and its delta in registers, so a row costs one pmullw and two
paddw with no reload of top[]. 16x16 and 32x32 run two groups together,
sharing the left[y] broadcast and writing a row with one 16 byte store.
Words suffice throughout: the largest value is 32162. SSSE3 rather than
SSE2 since pshufb does the broadcasts and the byte to word widening.

checkasm --bench on a Core Ultra 7 155H:
hevc_pred_planar_4x4_8_c:            18.4
hevc_pred_planar_4x4_8_ssse3:         5.5 ( 3.35x)
hevc_pred_planar_8x8_8_c:            83.1
hevc_pred_planar_8x8_8_ssse3:        16.3 ( 5.10x)
hevc_pred_planar_16x16_8_c:         224.8
hevc_pred_planar_16x16_8_ssse3:      47.3 ( 4.75x)
hevc_pred_planar_32x32_8_c:         629.5
hevc_pred_planar_32x32_8_ssse3:     196.5 ( 3.20x)

Part of #23022.
---
 libavcodec/hevc/pred.c          |   3 +
 libavcodec/hevc/pred.h          |   1 +
 libavcodec/x86/hevc/Makefile    |   2 +
 libavcodec/x86/hevc/pred.asm    | 205 ++++++++++++++++++++++++++++++++
 libavcodec/x86/hevc/pred_init.c |  51 ++++++++
 5 files changed, 262 insertions(+)
 create mode 100644 libavcodec/x86/hevc/pred.asm
 create mode 100644 libavcodec/x86/hevc/pred_init.c

diff --git a/libavcodec/hevc/pred.c b/libavcodec/hevc/pred.c
index 037cbc413f..673c02d856 100644
--- a/libavcodec/hevc/pred.c
+++ b/libavcodec/hevc/pred.c
@@ -85,4 +85,7 @@ void ff_hevc_pred_init(HEVCPredContext *hpc, int bit_depth)
 #if ARCH_MIPS
     ff_hevc_pred_init_mips(hpc, bit_depth);
 #endif
+#if ARCH_X86 && HAVE_X86ASM
+    ff_hevc_pred_init_x86(hpc, bit_depth);
+#endif
 }
diff --git a/libavcodec/hevc/pred.h b/libavcodec/hevc/pred.h
index 849806fefb..9c0cf17349 100644
--- a/libavcodec/hevc/pred.h
+++ b/libavcodec/hevc/pred.h
@@ -51,6 +51,7 @@ typedef struct HEVCPredContext {
 void ff_hevc_pred_init(HEVCPredContext *hpc, int bit_depth);
 void ff_hevc_pred_init_mips(HEVCPredContext *hpc, int bit_depth);
 void ff_hevc_pred_init_aarch64(HEVCPredContext *hpc, int bit_depth);
+void ff_hevc_pred_init_x86(HEVCPredContext *hpc, int bit_depth);
 
 /* C angular prediction fallbacks (non-static for arch-specific partial 
override) */
 #define HEVC_PRED_ANGULAR_DECL(depth)                                         \
diff --git a/libavcodec/x86/hevc/Makefile b/libavcodec/x86/hevc/Makefile
index d09c613a19..0ac5287bdc 100644
--- a/libavcodec/x86/hevc/Makefile
+++ b/libavcodec/x86/hevc/Makefile
@@ -7,6 +7,8 @@ X86ASM-OBJS-$(CONFIG_HEVC_DECODER)      += x86/hevc/dsp_init.o  
    \
                                            x86/hevc/dequant.o       \
                                            x86/hevc/idct.o          \
                                            x86/hevc/mc.o            \
+                                           x86/hevc/pred.o          \
+                                           x86/hevc/pred_init.o     \
                                            x86/hevc/sao.o           \
                                            x86/hevc/sao_10bit.o     \
                                            x86/h26x/h2656dsp.o      \
diff --git a/libavcodec/x86/hevc/pred.asm b/libavcodec/x86/hevc/pred.asm
new file mode 100644
index 0000000000..724a385a0c
--- /dev/null
+++ b/libavcodec/x86/hevc/pred.asm
@@ -0,0 +1,205 @@
+;******************************************************************************
+;* SIMD-optimized HEVC intra prediction
+;*
+;* This file is part of FFmpeg.
+;*
+;* FFmpeg is free software; you can redistribute it and/or
+;* modify it under the terms of the GNU Lesser General Public
+;* License as published by the Free Software Foundation; either
+;* version 2.1 of the License, or (at your option) any later version.
+;*
+;* FFmpeg is distributed in the hope that it will be useful,
+;* but WITHOUT ANY WARRANTY; without even the implied warranty of
+;* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
+;* Lesser General Public License for more details.
+;*
+;* You should have received a copy of the GNU Lesser General Public
+;* License along with FFmpeg; if not, write to the Free Software
+;* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
+;******************************************************************************
+
+%include "libavutil/x86/x86util.asm"
+
+SECTION_RODATA 16
+
+; planar weights: (size - 1 - x) is a suffix of pw_desc, (x + 1) a prefix of 
pw_asc
+pw_desc: dw 31, 30, 29, 28, 27, 26, 25, 24, 23, 22, 21, 20, 19, 18, 17, 16
+         dw 15, 14, 13, 12, 11, 10,  9,  8,  7,  6,  5,  4,  3,  2,  1,  0
+pw_asc:  dw  1,  2,  3,  4,  5,  6,  7,  8,  9, 10, 11, 12, 13, 14, 15, 16
+         dw 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32
+
+; pshufb masks: zero-extend bytes to words without a dedicated zero register
+pb_bcastw0: times 8 db 0, -1            ; byte 0 -> all eight words
+pb_widen:   db 0, -1, 1, -1, 2, -1, 3, -1, 4, -1, 5, -1, 6, -1, 7, -1
+
+; 4x4-specific: two rows per register
+pb_top4dup: db 0, -1, 1, -1, 2, -1, 3, -1, 0, -1, 1, -1, 2, -1, 3, -1
+pb_left01:  db 0, -1, 0, -1, 0, -1, 0, -1, 1, -1, 1, -1, 1, -1, 1, -1
+pb_left23:  db 2, -1, 2, -1, 2, -1, 2, -1, 3, -1, 3, -1, 3, -1, 3, -1
+pw_planar4_a:   dw 3, 2, 1, 0, 3, 2, 1, 0   ; size - 1 - x
+pw_planar4_b:   dw 1, 2, 3, 4, 1, 2, 3, 4   ; x + 1
+pw_planar4_vy0: dw 3, 3, 3, 3, 2, 2, 2, 2   ; size - 1 - y, rows 0-1
+pw_planar4_vy1: dw 1, 1, 1, 1, 0, 0, 0, 0   ; rows 2-3
+pw_planar4_yb0: dw 1, 1, 1, 1, 2, 2, 2, 2   ; y + 1, rows 0-1
+pw_planar4_yb1: dw 3, 3, 3, 3, 4, 4, 4, 4   ; rows 2-3
+
+cextern pw_4
+cextern pw_8
+cextern pw_16
+cextern pw_32
+
+SECTION .text
+
+; void ff_hevc_pred_planar_<idx>_8_<opt>(uint8_t *src, const uint8_t *top,
+;                                        const uint8_t *left, ptrdiff_t stride)
+;
+; src[y*stride + x] = ((size-1-x)*left[y] + (x+1)*top[size] +
+;                      (size-1-y)*top[x]  + (y+1)*left[size] + size) >> (log2 
+ 1)
+;
+; Everything except (size-1-x)*left[y] depends on y only through two terms that
+; are both linear in y, so for a fixed group of eight columns they collapse 
into
+; a single running accumulator
+;
+;   sum(y) = (x+1)*top[size] + (y+1)*left[size] + size + (size-1-y)*top[x]
+;   sum(y+1) - sum(y) = left[size] - top[x]
+;
+; which is loop-invariant. Iterating columns outermost and rows innermost keeps
+; sum and that delta in registers, so a row costs one pmullw and two paddw and
+; no reload of top[]. All intermediates stay in words: the largest value the
+; expression can take is 31*255 + 32*255 + 31*255 + 32*255 + 32 = 32162 < 2^15.
+
+; Set up the running accumulator for the eight columns starting at %6.
+; (size-1)*top[x] is formed as (top[x] << log2) - top[x] so that no vector of
+; size-1 has to be materialised.
+; %1 = sum, %2 = delta, %3 = temp, %4 = size, %5 = index, %6 = first column
+%macro PLANAR_INIT 6
+    movd            %1, [topq + %4]
+    pshufb          %1, [pb_bcastw0]        ; top[size]
+    pmullw          %1, [pw_asc + (%6) * 2] ; (x+1)*top[size]
+    movd            %2, [leftq + %4]
+    pshufb          %2, [pb_bcastw0]        ; left[size]
+    paddw           %1, %2
+    paddw           %1, [pw_ %+ %4]         ; + left[size] + size
+    movq            %3, [topq + (%6)]
+    pshufb          %3, [pb_widen]          ; top[x]
+    psubw           %2, %3                  ; delta = left[size] - top[x]
+    psubw           %1, %3
+    psllw           %3, (%5) + 2
+    paddw           %1, %3                  ; + (size-1)*top[x] = sum(0)
+%endmacro
+
+; One group of eight columns, for the block size that is only eight wide.
+; %1 = block size, %2 = index, %3 = first column
+%macro PLANAR_CHUNK 3
+    PLANAR_INIT     m0, m1, m2, %1, %2, %3
+    lea           srcq, [baseq + %3]
+    xor             yd, yd
+%%loop:
+    movd            m2, [leftq + yq]
+    pshufb          m2, [pb_bcastw0]                    ; left[y]
+    pmullw          m2, [pw_desc + (32 - %1 + %3) * 2]  ; (size-1-x)*left[y]
+    paddw           m2, m0
+    psrlw           m2, %2 + 3
+    packuswb        m2, m2
+    movq        [srcq], m2
+    add           srcq, strideq
+    paddw           m0, m1
+    inc             yd
+    cmp             yd, %1
+    jl %%loop
+%endmacro
+
+; Two adjacent groups at once, so that a row is written with a single 16 byte
+; store and the left[y] broadcast is shared between them.
+; %1 = block size, %2 = index, %3 = first column
+%macro PLANAR_CHUNK2 3
+    PLANAR_INIT     m0, m1, m4, %1, %2, %3
+    PLANAR_INIT     m2, m3, m4, %1, %2, %3 + 8
+    lea           srcq, [baseq + %3]
+    xor             yd, yd
+%%loop:
+    movd            m4, [leftq + yq]
+    pshufb          m4, [pb_bcastw0]                        ; left[y]
+    mova            m5, m4
+    pmullw          m4, [pw_desc + (32 - %1 + %3) * 2]
+    paddw           m4, m0
+    psrlw           m4, %2 + 3
+    pmullw          m5, [pw_desc + (32 - %1 + %3 + 8) * 2]
+    paddw           m5, m2
+    psrlw           m5, %2 + 3
+    packuswb        m4, m5
+    movu        [srcq], m4
+    add           srcq, strideq
+    paddw           m0, m1
+    paddw           m2, m3
+    inc             yd
+    cmp             yd, %1
+    jl %%loop
+%endmacro
+
+; %1 = block size, %2 = pred_planar index (log2(size) - 2)
+%macro PRED_PLANAR 2
+cglobal hevc_pred_planar_%2_8, 4, 6, 6, src, top, left, stride, base, y
+    mov          baseq, srcq
+%if %1 == 8
+    PLANAR_CHUNK %1, %2, 0
+%else
+%assign %%off 0
+%rep %1 / 16
+    PLANAR_CHUNK2 %1, %2, %%off
+%assign %%off %%off + 16
+%endrep
+%endif
+    RET
+%endmacro
+
+; 4x4: fully unrolled, two rows per register
+%macro PRED_PLANAR4 0
+cglobal hevc_pred_planar_0_8, 4, 4, 5, src, top, left, stride
+    movd            m0, [topq + 4]
+    pshufb          m0, [pb_bcastw0]        ; top[4]
+    pmullw          m0, [pw_planar4_b]      ; (x+1)*top[4]
+    mova            m1, m0
+    movd            m2, [leftq + 4]
+    pshufb          m2, [pb_bcastw0]        ; left[4]
+    mova            m3, m2
+    pmullw          m3, [pw_planar4_yb0]
+    paddw           m0, m3
+    paddw           m0, [pw_4]              ; row-invariant terms, rows 0-1
+    pmullw          m2, [pw_planar4_yb1]
+    paddw           m1, m2
+    paddw           m1, [pw_4]              ; rows 2-3
+    movd            m2, [topq]
+    pshufb          m2, [pb_top4dup]        ; top[0..3], both halves
+    movd            m3, [leftq]
+    mova            m4, m3
+    pshufb          m3, [pb_left01]         ; left[0] x4 | left[1] x4
+    pshufb          m4, [pb_left23]         ; left[2] x4 | left[3] x4
+    pmullw          m3, [pw_planar4_a]
+    paddw           m3, m0
+    mova            m0, m2
+    pmullw          m0, [pw_planar4_vy0]
+    paddw           m3, m0
+    psrlw           m3, 3
+    pmullw          m4, [pw_planar4_a]
+    paddw           m4, m1
+    pmullw          m2, [pw_planar4_vy1]
+    paddw           m4, m2
+    psrlw           m4, 3
+    packuswb        m3, m4
+    movd        [srcq], m3
+    psrldq          m3, 4
+    movd [srcq + strideq], m3
+    lea           srcq, [srcq + strideq * 2]
+    psrldq          m3, 4
+    movd        [srcq], m3
+    psrldq          m3, 4
+    movd [srcq + strideq], m3
+    RET
+%endmacro
+
+INIT_XMM ssse3
+PRED_PLANAR4
+PRED_PLANAR  8, 1
+PRED_PLANAR 16, 2
+PRED_PLANAR 32, 3
diff --git a/libavcodec/x86/hevc/pred_init.c b/libavcodec/x86/hevc/pred_init.c
new file mode 100644
index 0000000000..980e28b861
--- /dev/null
+++ b/libavcodec/x86/hevc/pred_init.c
@@ -0,0 +1,51 @@
+/*
+ * SIMD-optimized HEVC intra prediction
+ *
+ * This file is part of FFmpeg.
+ *
+ * FFmpeg is free software; you can redistribute it and/or
+ * modify it under the terms of the GNU Lesser General Public
+ * License as published by the Free Software Foundation; either
+ * version 2.1 of the License, or (at your option) any later version.
+ *
+ * FFmpeg is distributed in the hope that it will be useful,
+ * but WITHOUT ANY WARRANTY; without even the implied warranty of
+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
+ * Lesser General Public License for more details.
+ *
+ * You should have received a copy of the GNU Lesser General Public
+ * License along with FFmpeg; if not, write to the Free Software
+ * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
+ */
+
+#include "config.h"
+
+#include "libavutil/attributes.h"
+#include "libavutil/cpu.h"
+#include "libavutil/x86/cpu.h"
+#include "libavcodec/hevc/pred.h"
+
+#define PRED_PLANAR_FUNC(idx, opt)                                            \
+void ff_hevc_pred_planar_ ## idx ## _8_ ## opt(uint8_t *src,                  \
+                                               const uint8_t *top,            \
+                                               const uint8_t *left,           \
+                                               ptrdiff_t stride);
+
+PRED_PLANAR_FUNC(0, ssse3)
+PRED_PLANAR_FUNC(1, ssse3)
+PRED_PLANAR_FUNC(2, ssse3)
+PRED_PLANAR_FUNC(3, ssse3)
+
+av_cold void ff_hevc_pred_init_x86(HEVCPredContext *hpc, int bit_depth)
+{
+    int cpu_flags = av_get_cpu_flags();
+
+    if (bit_depth == 8) {
+        if (EXTERNAL_SSSE3(cpu_flags)) {
+            hpc->pred_planar[0] = ff_hevc_pred_planar_0_8_ssse3;
+            hpc->pred_planar[1] = ff_hevc_pred_planar_1_8_ssse3;
+            hpc->pred_planar[2] = ff_hevc_pred_planar_2_8_ssse3;
+            hpc->pred_planar[3] = ff_hevc_pred_planar_3_8_ssse3;
+        }
+    }
+}
-- 
2.52.0

_______________________________________________
ffmpeg-devel mailing list -- [email protected]
To unsubscribe send an email to [email protected]

Reply via email to