This is an automated email from the git hooks/post-receive script. Git pushed a commit to branch master in repository ffmpeg.
commit 357bb4965f276294ef051f06e7ebebc813e46f64 Author: Lynne <[email protected]> AuthorDate: Thu Jun 4 15:04:13 2026 +0900 Commit: Lynne <[email protected]> CommitDate: Sun Oct 4 17:38:35 2026 +0900 lavu/tx: add AArch64 NEON fft_pfa_15xM Same as the x86 version. --- libavutil/aarch64/tx_float_init.c | 77 ++++++++++++++++++++++++------ libavutil/aarch64/tx_float_neon.S | 99 +++++++++++++++++++++++++++++++++++++++ 2 files changed, 162 insertions(+), 14 deletions(-) diff --git a/libavutil/aarch64/tx_float_init.c b/libavutil/aarch64/tx_float_init.c index 47f1e12700..a049562609 100644 --- a/libavutil/aarch64/tx_float_init.c +++ b/libavutil/aarch64/tx_float_init.c @@ -19,6 +19,7 @@ #define TX_FLOAT #include "libavutil/tx_priv.h" #include "libavutil/attributes.h" +#include "libavutil/mem.h" #include "libavutil/aarch64/cpu.h" TX_DECL_FN(fft2, neon) @@ -34,6 +35,8 @@ TX_DECL_FN(fft32, neon) TX_DECL_FN(fft32_ns, neon) TX_DECL_FN(fft_sr, neon) TX_DECL_FN(fft_sr_ns, neon) +TX_DECL_FN(fft_pfa_15xM, neon) +TX_DECL_FN(fft_pfa_15xM_ns, neon) static av_cold int neon_init(AVTXContext *s, const FFTXCodelet *cd, uint64_t flags, FFTXCodeletOptions *opts, @@ -46,11 +49,29 @@ static av_cold int neon_init(AVTXContext *s, const FFTXCodelet *cd, return ff_tx_gen_split_radix_parity_revtab(s, len, inv, opts, 8, 0); } +/* Reorder one 15-point map so the loads in the pre-permuted assembly path + * become simple contiguous chunks. Mirrors the x86 FFT15 init. */ +static void fft15_permute_map(int *map) +{ + int cnt = 0, tmp[15]; + memcpy(tmp, map, 15*sizeof(*tmp)); + for (int i = 1; i < 15; i += 3) + map[cnt++] = tmp[i]; + for (int i = 2; i < 15; i += 3) + map[cnt++] = tmp[i]; + for (int i = 0; i < 15; i += 3) + map[cnt++] = tmp[i]; + memmove(&map[7], &map[6], 4*sizeof(int)); + memmove(&map[3], &map[1], 4*sizeof(int)); + map[1] = tmp[2]; + map[2] = tmp[0]; +} + static av_cold int fft15_init(AVTXContext *s, const FFTXCodelet *cd, uint64_t flags, FFTXCodeletOptions *opts, int len, int inv, const void *scale) { - int ret, cnt = 0, tmp[15]; + int ret; FFTXCodeletOptions sub_opts = { .map_dir = FF_TX_MAP_GATHER }; ff_tx_init_tabs_float(len); @@ -58,19 +79,44 @@ static av_cold int fft15_init(AVTXContext *s, const FFTXCodelet *cd, if ((ret = ff_tx_gen_pfa_input_map(s, &sub_opts, 3, 5)) < 0) return ret; - /* Reorder the 15-pt map so the loads in the pre-permuted assembly path - * become simple contiguous chunks. Mirrors the x86 FFT15 init. */ - memcpy(tmp, s->map, 15*sizeof(*tmp)); - for (int i = 1; i < 15; i += 3) - s->map[cnt++] = tmp[i]; - for (int i = 2; i < 15; i += 3) - s->map[cnt++] = tmp[i]; - for (int i = 0; i < 15; i += 3) - s->map[cnt++] = tmp[i]; - memmove(&s->map[7], &s->map[6], 4*sizeof(int)); - memmove(&s->map[3], &s->map[1], 4*sizeof(int)); - s->map[1] = tmp[2]; - s->map[2] = tmp[0]; + fft15_permute_map(s->map); + + return 0; +} + +/* 15xM prime-factor FFT: M inlined 15-point transforms followed by 15 calls to + * a power-of-two subtransform. Mirrors the x86 fft_pfa_15xM, but the aarch64 + * ABI lets us call the subtransform normally, so no FF_TX_ASM_CALL is needed. */ +static av_cold int fft_pfa_init(AVTXContext *s, const FFTXCodelet *cd, + uint64_t flags, FFTXCodeletOptions *opts, + int len, int inv, const void *scale) +{ + int ret; + int sub_len = len / cd->factors[0]; + FFTXCodeletOptions sub_opts = { .map_dir = FF_TX_MAP_SCATTER }; + + flags &= ~FF_TX_OUT_OF_PLACE; /* We want the subtransform to be */ + flags |= AV_TX_INPLACE; /* in-place */ + flags |= FF_TX_PRESHUFFLE; /* This function handles the permute step */ + + if ((ret = ff_tx_init_subtx(s, TX_TYPE(FFT), flags, &sub_opts, + sub_len, inv, scale))) + return ret; + + if ((ret = ff_tx_gen_compound_mapping(s, opts, s->inv, + cd->factors[0], sub_len))) + return ret; + + /* The 15-point transform is itself a compound one, so embed its input map + * and apply the same load-friendly reorder used by fft15_init. */ + TX_EMBED_INPUT_PFA_MAP(s->map, len, 3, 5); + for (int k = 0; k < sub_len; k++) + fft15_permute_map(&s->map[k*15]); + + if (!(s->tmp = av_malloc(len*sizeof(*s->tmp)))) + return AVERROR(ENOMEM); + + ff_tx_init_tabs_float(len / sub_len); return 0; } @@ -93,5 +139,8 @@ const FFTXCodelet * const ff_tx_codelet_list_float_aarch64[] = { TX_DEF(fft_sr, FFT, 64, 131072, 2, 0, 128, neon_init, neon, NEON, 0, 0), TX_DEF(fft_sr_ns, FFT, 64, 131072, 2, 0, 192, neon_init, neon, NEON, AV_TX_INPLACE | FF_TX_PRESHUFFLE, 0), + TX_DEF(fft_pfa_15xM, FFT, 60, TX_LEN_UNLIMITED, 15, 2, 128, fft_pfa_init, neon, NEON, AV_TX_INPLACE, 0), + TX_DEF(fft_pfa_15xM_ns, FFT, 60, TX_LEN_UNLIMITED, 15, 2, 192, fft_pfa_init, neon, NEON, AV_TX_INPLACE | FF_TX_PRESHUFFLE, 0), + NULL, }; diff --git a/libavutil/aarch64/tx_float_neon.S b/libavutil/aarch64/tx_float_neon.S index 255dc3acaa..2db0e95740 100644 --- a/libavutil/aarch64/tx_float_neon.S +++ b/libavutil/aarch64/tx_float_neon.S @@ -671,6 +671,105 @@ endfunc FFT15_FN float, 0 FFT15_FN ns_float, 1 +// 15xM PFA (len = 15*M, M a power of two), like the x86 fft_pfa_15xM: +// dim1 = M 15pt transforms scattered into s->tmp at sub_map[i], spaced M +// apart; dim2 = 15 in-place M-pt subtransforms (plain blr, the uniform ABI +// needs no asm-call variant); post = out[i*stride] = s->tmp[out_map[i]]. +// AVTXContext offsets: len=0, map=8, tmp=24, sub=32, fn[0]=40. +.macro PFA_15_FN name, no_perm +function ff_tx_fft_pfa_15xM_\name\()_neon, export=1 + AARCH64_SIGN_LINK_REGISTER + stp x29, x30, [sp, #-128]! + mov x29, sp + stp x19, x20, [sp, #16] + stp x21, x22, [sp, #32] + stp x23, x24, [sp, #48] + stp x25, x26, [sp, #64] + stp x27, x28, [sp, #80] + stp d8, d9, [sp, #96] + str d10, [sp, #112] + + mov x25, x0 // root context + mov x26, x1 // user out + mov x27, x3 // user stride (bytes) + + ldr w19, [x0, #0] // len = 15*M + ldr x20, [x0, #24] // s->tmp + ldr x23, [x0, #32] // s->sub + ldr w24, [x23, #0] // M + ldr x28, [x23, #8] // sub_map + lsl x3, x24, #3 // M*8 = 15pt output stride +.if \no_perm == 0 + ldr x4, [x0, #8] // in_map, advanced by the loads +.endif + + movrel x5, X(ff_tx_tab_53_float) + ld1 { v28.4s, v29.4s, v30.4s }, [x5] + movrel x5, tab_15pt + ld1 { v24.4s }, [x5] + LOAD_SUBADD + FFT15_DERIVE_CONSTS + add x6, x3, x3, lsl #1 // stride*3 + add x7, x3, x3, lsl #2 // stride*5 +1: + ldr w10, [x28], #4 // sub_map[i] + add x1, x20, x10, lsl #3 + FFT15_LOAD \no_perm, advance=1 + FFT15_CORE hoist_strides=1 + subs w19, w19, #15 + b.gt 1b + + ldr x28, [x25, #40] // ctx->fn[0] + mov x21, x20 // column base + mov w19, #15 +2: + mov x0, x23 + mov x1, x21 + mov x2, x21 + mov x3, #8 + blr x28 // M-point FFT, in-place + add x21, x21, x24, lsl #3 + subs w19, w19, #1 + b.gt 2b + + // out[i*stride] = s->tmp[out_map[i]], unrolled by 4 (len % 60 == 0) + ldr w19, [x25, #0] // len + ldr x22, [x25, #8] // s->map + add x22, x22, x19, lsl #2 // out_map = map + len + mov x10, x26 + lsl x11, x27, #1 // 2*stride + add x12, x27, x11 // 3*stride +3: + ldp w13, w14, [x22], #8 // out_map[i, i+1] + ldp w15, w16, [x22], #8 // out_map[i+2, i+3] + ldr d0, [x20, x13, lsl #3] + ldr d1, [x20, x14, lsl #3] + ldr d2, [x20, x15, lsl #3] + ldr d3, [x20, x16, lsl #3] + str d0, [x10] + str d1, [x10, x27] + str d2, [x10, x11] + str d3, [x10, x12] + add x10, x10, x11, lsl #1 + subs w19, w19, #4 + b.gt 3b + + ldp x19, x20, [sp, #16] + ldp x21, x22, [sp, #32] + ldp x23, x24, [sp, #48] + ldp x25, x26, [sp, #64] + ldp x27, x28, [sp, #80] + ldp d8, d9, [sp, #96] + ldr d10, [sp, #112] + ldp x29, x30, [sp], #128 + AARCH64_VALIDATE_LINK_REGISTER + ret +endfunc +.endm + +PFA_15_FN float, 0 +PFA_15_FN ns_float, 1 + .macro SETUP_SR_RECOMB len, re, im, dec ldr w5, =(\len - 4*7) movrel \re, X(ff_tx_tab_\len\()_float) -- To stop receiving notification emails like this one, please contact [email protected]. _______________________________________________ ffmpeg-cvslog mailing list -- [email protected] To unsubscribe send an email to [email protected]
