Hi,

Gentle ping on this series. It adds an AVX2 implementation of the VVC
planar intra prediction kernel (8-bit), with a checkasm test. It still
applies cleanly to current master and passes checkasm and fate-vvc.

I'd appreciate any review or feedback when someone has time.

Thanks,
Samith

On Sun, Jul 26, 2026 at 3:07 AM Gudikandula Samith
<[email protected]> wrote:
>
> Implements pred_planar for x86-64 AVX2 at 8 bit.
>
> Benchmarks on AMD Ryzen 7 5825U:
>   vvc_pred_planar_4x4_8_c:         52.8
>   vvc_pred_planar_4x4_8_avx2:      24.1 (2.16x)
>   vvc_pred_planar_8x8_8_c:        208.3
>   vvc_pred_planar_8x8_8_avx2:      48.3 (4.31x)
>   vvc_pred_planar_16x16_8_c:      778.0
>   vvc_pred_planar_16x16_8_avx2:   160.4 (4.80x)
>   vvc_pred_planar_32x32_8_c:     2356.0
>   vvc_pred_planar_32x32_8_avx2:   580.3 (4.02x)
>   vvc_pred_planar_64x64_8_c:     7414.0
>   vvc_pred_planar_64x64_8_avx2:  2185.9 (3.27x)
>
> Signed-off-by: Gudikandula Samith <[email protected]>
> ---
>  libavcodec/x86/vvc/Makefile   |   1 +
>  libavcodec/x86/vvc/dsp_init.c |  13 ++
>  libavcodec/x86/vvc/intra.asm  | 224 ++++++++++++++++++++++++++++++++++
>  3 files changed, 238 insertions(+)
>  create mode 100644 libavcodec/x86/vvc/intra.asm
>
> diff --git a/libavcodec/x86/vvc/Makefile b/libavcodec/x86/vvc/Makefile
> index 0cebfb4e9e..b24ea6e692 100644
> --- a/libavcodec/x86/vvc/Makefile
> +++ b/libavcodec/x86/vvc/Makefile
> @@ -4,6 +4,7 @@ clean::
>  X86ASM-OBJS-$(CONFIG_VVC_DECODER)      += x86/vvc/dsp_init.o        \
>                                            x86/vvc/alf.o             \
>                                            x86/vvc/dmvr.o            \
> +                                          x86/vvc/intra.o           \
>                                            x86/vvc/mc.o              \
>                                            x86/vvc/of.o              \
>                                            x86/vvc/sad.o             \
> diff --git a/libavcodec/x86/vvc/dsp_init.c b/libavcodec/x86/vvc/dsp_init.c
> index 6802294795..4f187e1eb1 100644
> --- a/libavcodec/x86/vvc/dsp_init.c
> +++ b/libavcodec/x86/vvc/dsp_init.c
> @@ -292,6 +292,16 @@ void bf(ff_vvc_alf_filter_chroma, bd, opt)(uint8_t *dst, 
> ptrdiff_t dst_stride, \
>      c->alf.classify       = bf(vvc_alf_classify, bd, opt);                   
>   \
>  } while (0)
>
> +#define INTRA_PROTOTYPE(bd, opt)                                             
>   \
> +void ff_vvc_pred_planar_##bd##_##opt(uint8_t *src, const uint8_t *top,       
>   \
> +    const uint8_t *left, int w, int h, ptrdiff_t stride);
> +
> +INTRA_PROTOTYPE(8, avx2)
> +
> +#define INTRA_INIT(bd, opt) do {                                             
>   \
> +    c->intra.pred_planar = ff_vvc_pred_planar_##bd##_##opt;                  
>   \
> +} while (0)
> +
>  #endif
>
>
> @@ -321,6 +331,9 @@ av_cold void ff_vvc_dsp_init_x86(VVCDSPContext *const c, 
> const int bd)
>              // filter
>              ALF_INIT(8, avx2);
>              SAO_INIT(8, avx2);
> +
> +            // intra
> +            INTRA_INIT(8, avx2);
>          }
>  #endif
>          break;
> diff --git a/libavcodec/x86/vvc/intra.asm b/libavcodec/x86/vvc/intra.asm
> new file mode 100644
> index 0000000000..c18c17193a
> --- /dev/null
> +++ b/libavcodec/x86/vvc/intra.asm
> @@ -0,0 +1,224 @@
> +; /*
> +; * Provide SIMD planar intra prediction functions for VVC decoding
> +; *
> +; * Copyright (c) 2026 Gudikandula Samith <[email protected]>
> +; *
> +; * This file is part of FFmpeg.
> +; *
> +; * FFmpeg is free software; you can redistribute it and/or
> +; * modify it under the terms of the GNU Lesser General Public
> +; * License as published by the Free Software Foundation; either
> +; * version 2.1 of the License, or (at your option) any later version.
> +; *
> +; * FFmpeg is distributed in the hope that it will be useful,
> +; * but WITHOUT ANY WARRANTY; without even the implied warranty of
> +; * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
> +; * Lesser General Public License for more details.
> +; *
> +; * You should have received a copy of the GNU Lesser General Public
> +; * License along with FFmpeg; if not, write to the Free Software
> +; * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 
> USA
> +; */
> +
> +%include "libavutil/x86/x86util.asm"
> +
> +SECTION_RODATA 32
> +
> +pw_wm1x_w4:  dw  3,  2,  1,  0,  0,  0,  0,  0
> +pw_xp1_w4:   dw  1,  2,  3,  4,  0,  0,  0,  0
> +pw_wm1x_w8:  dw  7,  6,  5,  4,  3,  2,  1,  0
> +pw_xp1_w8:   dw  1,  2,  3,  4,  5,  6,  7,  8
> +pw_wm1x_w16: dw  15, 14, 13, 12, 11, 10,  9,  8,  7,  6,  5,  4,  3,  2,  1, 
>  0  ; (w - 1 - x)
> +pw_xp1_w16:  dw  1,  2,  3,  4,  5,  6,  7,  8,  9, 10, 11, 12, 13, 14, 15, 
> 16  ; (x + 1)
> +pw_wm1x_w32: dw  31, 30, 29, 28, 27, 26, 25, 24, 23, 22, 21, 20, 19, 18, 17, 
> 16, \
> +                 15, 14, 13, 12, 11, 10, 9,  8,  7,  6,  5,  4,  3,  2,  1,  > 0
> +pw_xp1_w32:  dw  1,  2,  3,  4,  5,  6,  7,  8, 9,  10, 11, 12, 13, 14, 15, 
> 16, \
> +                 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 
> 32
> +pw_wm1x_w64: dw  63, 62, 61, 60, 59, 58, 57, 56, 55, 54, 53, 52, 51, 50, 49, 
> 48, \
> +                 47, 46, 45, 44, 43, 42, 41, 40, 39, 38, 37, 36, 35, 34, 33, 
> 32, \
> +                 31, 30, 29, 28, 27, 26, 25, 24, 23, 22, 21, 20, 19, 18, 17, 
> 16, \
> +                 15, 14, 13, 12, 11, 10, 9,  8,  7,  6,  5,  4,  3,  2,  1,  > 0
> +pw_xp1_w64:  dw  1,  2,  3,  4,  5,  6,  7,  8,  9,  10, 11, 12, 13, 14, 15, 
> 16, \
> +                 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 
> 32, \
> +                 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 
> 48, \
> +                 49, 50, 51, 52, 53, 54, 55, 56, 57, 58, 59, 60, 61, 62, 63, 
> 64
> +
> +
> +SECTION .text
> +
> +%if ARCH_X86_64
> +%if HAVE_AVX2_EXTERNAL
> +
> +INIT_YMM avx2
> +
> +; Per-row scalars, broadcast into vector registers:
> +; m8 = A = (h-1-y)
> +; m9 = B = (y+1) * left[h]
> +; m10 = C = left[y]
> +%macro ROW_CONSTANTS 0
> +    mov           eax, r12d
> +    sub           eax, r11d           ;A=(h-1-y)
> +    movd          xm8, eax
> +    vpbroadcastw   m8, xm8
> +    lea            eax, [r11d+1]
> +    imul           eax, r10d          ;B=(y+1)*left[h]
> +    movd           xm9, eax
> +    vpbroadcastw    m9, xm9
> +    movzx           eax, byte [leftq+r11]
> +    movd           xm10, eax
> +    vpbroadcastw   m10, xm10
> +%endmacro
> +; Block-constant setup shared by every width.
> +;  %1 = logw (immediate), %2 = top[w] byte offset (=width)
> +%macro PLANAR_SETUP 2
> +    mov            r13d, hd
> +    bsr             r6d, hd
> +    lea             r7d, [r6d +%1 +1]
> +    mov             r8d, hd
> +    shl             r8d, %1          ;size=w*h= h<<logw
> +    movzx           r9d, byte [topq + %2]
> +    movzx          r10d, byte [leftq +r13]
> +    movd            xm3, r9d
> +    vpbroadcastw     m3, xm3
> +    movd            xm4, r8d
> +    vpbroadcastd     m4, xm4
> +    movd            xm5, r6d
> +    movd             xm6, r7d
> +    mov             r12d, hd
> +    dec             r12d
> +    xor             r11d, r11d
> +%endmacro
> +
> +; Process 16 columns starting at column offset %3.
> +;   %1 = (w-1-x) table, %2 = (x+1) table, %3 = column offset, %4 = logw 
> immediate
> +%macro PLANAR16 4
> +    pmovzxbw         m0, [topq+%3]
> +    movu             m1, [%1 + %3*2]
> +    movu             m2, [%2 + %3*2]
> +    vpmullw          m11, m0, m8
> +    vpaddw           m11, m11, m9
> +    vpmullw          m12, m1, m10
> +    vpmullw           m2, m2, m3
> +    vpaddw           m12, m12, m2
> +    pmovzxwd         m13, xm11
> +    pmovzxwd         m14, xm12
> +    vpslld           m13, m13, %4
> +    vpslld           m14, m14, xm5
> +    vpaddd           m13, m13, m14
> +    vpaddd           m13, m13, m4
> +    vpsrad           m13, m13, xm6
> +    vextracti128     xm11, m11, 1
> +    vextracti128     xm12, m12, 1
> +    pmovzxwd         m14, xm11
> +    pmovzxwd         m15, xm12
> +    vpslld           m14, m14, %4
> +    vpslld           m15, m15, xm5
> +    vpaddd           m14, m14, m15
> +    vpaddd           m14, m14, m4
> +    vpsrad           m14, m14, xm6
> +    packusdw         m13, m13, m14
> +    vpermq           m13, m13, 11011000b
> +    packuswb         m13, m13, m13
> +    vpermq           m13, m13, 11011000b
> +    movu   [srcq + %3], xm13
> +%endmacro
> +
> +; Process a narrow row of 4 or 8 columns (offset 0).
> +;   %1 = (w-1-x) table, %2 = (x+1) table, %3 = logw immediate, %4 = store 
> instruction
> +%macro PLANAR_NARROW 4
> +    pmovzxbw          xm0, [topq]
> +    movu              xm1, [%1]
> +    movu              xm2, [%2]
> +    vpmullw           xm11, xm0, xm8
> +    vpaddw            xm11, xm11, xm9
> +    vpmullw           xm12, xm1, xm10
> +    vpmullw           xm2, xm2, xm3
> +    vpaddw            xm12, xm12, xm2
> +    pmovzxwd          m13, xm11
> +    pmovzxwd          m14, xm12
> +    vpslld            m13, m13, %3
> +    vpslld            m14, m14, xm5
> +    vpaddd            m13, m13, m14
> +    vpaddd            m13, m13, m4
> +    vpsrad            m13, m13, xm6
> +    packusdw          m13, m13, m13
> +    vpermq            m13, m13, 11011000b
> +    packuswb          m13, m13, m13
> +    %4      [srcq], xm13
> +%endmacro
> +
> +; void ff_vvc_pred_planar_8_avx2(uint8_t *src, const uint8_t *top,
> +;                                const uint8_t *left, int w, int h,
> +;                                ptrdiff_t stride);
> +cglobal vvc_pred_planar_8, 6, 14, 16, src, top, left, w, h, stride
> +    cmp             wd, 16
> +    je .w16
> +    cmp             wd, 8
> +    je .w8
> +    cmp             wd, 32
> +    je .w32
> +    cmp             wd, 64
> +    je .w64
> +    ; fall through to width 4
> +
> +.w4:
> +    PLANAR_SETUP 2, 4
> +.w4_row:
> +    ROW_CONSTANTS
> +    PLANAR_NARROW pw_wm1x_w4, pw_xp1_w4, 2, movd
> +    add            srcq, strideq
> +    inc            r11d
> +    cmp            r11d, hd
> +    jl .w4_row
> +    RET
> +
> +.w8:
> +    PLANAR_SETUP 3, 8
> +.w8_row:
> +    ROW_CONSTANTS
> +    PLANAR_NARROW pw_wm1x_w8, pw_xp1_w8, 3, movq
> +    add            srcq, strideq
> +    inc            r11d
> +    cmp            r11d, hd
> +    jl .w8_row
> +    RET
> +
> +.w16:
> +    PLANAR_SETUP 4, 16
> +.w16_row:
> +    ROW_CONSTANTS
> +    PLANAR16 pw_wm1x_w16, pw_xp1_w16, 0, 4
> +    add            srcq, strideq
> +    inc            r11d
> +    cmp            r11d, hd
> +    jl .w16_row
> +    RET
> +
> +.w32:
> +    PLANAR_SETUP 5, 32
> +.w32_row:
> +    ROW_CONSTANTS
> +    PLANAR16 pw_wm1x_w32, pw_xp1_w32, 0,  5
> +    PLANAR16 pw_wm1x_w32, pw_xp1_w32, 16, 5
> +    add            srcq, strideq
> +    inc            r11d
> +    cmp            r11d, hd
> +    jl .w32_row
> +    RET
> +
> +.w64:
> +    PLANAR_SETUP 6, 64
> +.w64_row:
> +    ROW_CONSTANTS
> +    PLANAR16 pw_wm1x_w64, pw_xp1_w64, 0,  6
> +    PLANAR16 pw_wm1x_w64, pw_xp1_w64, 16, 6
> +    PLANAR16 pw_wm1x_w64, pw_xp1_w64, 32, 6
> +    PLANAR16 pw_wm1x_w64, pw_xp1_w64, 48, 6
> +    add            srcq, strideq
> +    inc            r11d
> +    cmp            r11d, hd
> +    jl .w64_row
> +    RET
> +
> +%endif
> +%endif
> --
> 2.53.0
>
_______________________________________________
ffmpeg-devel mailing list -- [email protected]
To unsubscribe send an email to [email protected]

Reply via email to