This is an automated email from the git hooks/post-receive script. Git pushed a commit to branch master in repository ffmpeg.
commit 31f5c05c4e0a1422eff9c87358c25488e0104278 Author: Lynne <[email protected]> AuthorDate: Thu Jun 4 06:03:55 2026 +0900 Commit: Lynne <[email protected]> CommitDate: Sun Oct 4 17:38:35 2026 +0900 lavu/tx: add AArch64 NEON fft15 codelet Based on the C code in doc/transforms.md. --- libavutil/aarch64/tx_float_init.c | 33 ++++++ libavutil/aarch64/tx_float_neon.S | 233 ++++++++++++++++++++++++++++++++++++++ 2 files changed, 266 insertions(+) diff --git a/libavutil/aarch64/tx_float_init.c b/libavutil/aarch64/tx_float_init.c index 8300472c4c..47f1e12700 100644 --- a/libavutil/aarch64/tx_float_init.c +++ b/libavutil/aarch64/tx_float_init.c @@ -26,6 +26,8 @@ TX_DECL_FN(fft4_fwd, neon) TX_DECL_FN(fft4_inv, neon) TX_DECL_FN(fft8, neon) TX_DECL_FN(fft8_ns, neon) +TX_DECL_FN(fft15, neon) +TX_DECL_FN(fft15_ns, neon) TX_DECL_FN(fft16, neon) TX_DECL_FN(fft16_ns, neon) TX_DECL_FN(fft32, neon) @@ -44,6 +46,35 @@ static av_cold int neon_init(AVTXContext *s, const FFTXCodelet *cd, return ff_tx_gen_split_radix_parity_revtab(s, len, inv, opts, 8, 0); } +static av_cold int fft15_init(AVTXContext *s, const FFTXCodelet *cd, + uint64_t flags, FFTXCodeletOptions *opts, + int len, int inv, const void *scale) +{ + int ret, cnt = 0, tmp[15]; + FFTXCodeletOptions sub_opts = { .map_dir = FF_TX_MAP_GATHER }; + + ff_tx_init_tabs_float(len); + + if ((ret = ff_tx_gen_pfa_input_map(s, &sub_opts, 3, 5)) < 0) + return ret; + + /* Reorder the 15-pt map so the loads in the pre-permuted assembly path + * become simple contiguous chunks. Mirrors the x86 FFT15 init. */ + memcpy(tmp, s->map, 15*sizeof(*tmp)); + for (int i = 1; i < 15; i += 3) + s->map[cnt++] = tmp[i]; + for (int i = 2; i < 15; i += 3) + s->map[cnt++] = tmp[i]; + for (int i = 0; i < 15; i += 3) + s->map[cnt++] = tmp[i]; + memmove(&s->map[7], &s->map[6], 4*sizeof(int)); + memmove(&s->map[3], &s->map[1], 4*sizeof(int)); + s->map[1] = tmp[2]; + s->map[2] = tmp[0]; + + return 0; +} + const FFTXCodelet * const ff_tx_codelet_list_float_aarch64[] = { TX_DEF(fft2, FFT, 2, 2, 2, 0, 128, NULL, neon, NEON, AV_TX_INPLACE, 0), TX_DEF(fft2, FFT, 2, 2, 2, 0, 192, neon_init, neon, NEON, AV_TX_INPLACE | FF_TX_PRESHUFFLE, 0), @@ -52,6 +83,8 @@ const FFTXCodelet * const ff_tx_codelet_list_float_aarch64[] = { TX_DEF(fft4_inv, FFT, 4, 4, 2, 0, 128, NULL, neon, NEON, AV_TX_INPLACE | FF_TX_INVERSE_ONLY, 0), TX_DEF(fft8, FFT, 8, 8, 2, 0, 128, neon_init, neon, NEON, AV_TX_INPLACE, 0), TX_DEF(fft8_ns, FFT, 8, 8, 2, 0, 192, neon_init, neon, NEON, AV_TX_INPLACE | FF_TX_PRESHUFFLE, 0), + TX_DEF(fft15, FFT, 15, 15, 15, 0, 128, fft15_init, neon, NEON, AV_TX_INPLACE, 0), + TX_DEF(fft15_ns, FFT, 15, 15, 15, 0, 192, fft15_init, neon, NEON, AV_TX_INPLACE | FF_TX_PRESHUFFLE, 0), TX_DEF(fft16, FFT, 16, 16, 2, 0, 128, neon_init, neon, NEON, AV_TX_INPLACE, 0), TX_DEF(fft16_ns, FFT, 16, 16, 2, 0, 192, neon_init, neon, NEON, AV_TX_INPLACE | FF_TX_PRESHUFFLE, 0), TX_DEF(fft32, FFT, 32, 32, 2, 0, 128, neon_init, neon, NEON, AV_TX_INPLACE, 0), diff --git a/libavutil/aarch64/tx_float_neon.S b/libavutil/aarch64/tx_float_neon.S index 12c4e880dc..255dc3acaa 100644 --- a/libavutil/aarch64/tx_float_neon.S +++ b/libavutil/aarch64/tx_float_neon.S @@ -438,6 +438,239 @@ endfunc FFT16_FN float, 0 FFT16_FN ns_float, 1 +const tab_15pt, align=4 + .float 1.0, 1.0, -1.0, -1.0 +endconst + +// Tab_53 twiddles (v28..v30) duplicated/pre-signed once, instead of per +// transform: v8/v9 = -+tab[8,9]/[10,11], v25/v28/v29 = tab[0,1]/[2,3]/[4,5], +// v10 = +-tab[6,7]. v30 keeps tab[8..11]. Callers preserve d8-d10. +.macro FFT15_DERIVE_CONSTS + dup v8.2d, v30.d[0] + dup v9.2d, v30.d[1] + dup v10.2d, v29.d[1] + dup v25.2d, v28.d[0] + dup v28.2d, v28.d[1] + dup v29.2d, v29.d[0] + fmul v8.4s, v8.4s, v31.4s + fmul v9.4s, v9.4s, v31.4s + fmul v10.4s, v10.4s, v24.4s +.endm + +.macro FFT15_LOAD no_perm, advance=0 +.if \no_perm == 1 + // Writebacks leave x2 a whole transform (120B) ahead for the PFA loop + ld1 { v0.4s }, [x2], #16 // in[0,1] + ld1r { v1.2d }, [x2], #8 // in[2] duplicated + ld1 { v2.4s, v3.4s, v4.4s }, [x2], #48 // in[3..8] + ld1 { v5.4s, v6.4s, v7.4s }, [x2], #48 // in[9..14] +.else + ldp w10, w11, [x4] // lut[0,1] + ldr w12, [x4, #8] // lut[2] + ldp w13, w14, [x4, #12] // lut[3,4] + ldp w15, w16, [x4, #20] // lut[5,6] + + ldr d0, [x2, x10, lsl #3] + add x10, x2, x11, lsl #3 + add x12, x2, x12, lsl #3 + ld1 { v0.d }[1], [x10] + ld1r { v1.2d }, [x12] + + ldr d2, [x2, x13, lsl #3] + add x13, x2, x14, lsl #3 + ldr d3, [x2, x15, lsl #3] + add x15, x2, x16, lsl #3 + ld1 { v2.d }[1], [x13] + ld1 { v3.d }[1], [x15] + + ldp w10, w11, [x4, #28] // lut[7,8] + ldp w12, w13, [x4, #36] // lut[9,10] + ldp w14, w15, [x4, #44] // lut[11,12] + ldp w16, w17, [x4, #52] // lut[13,14] +.if \advance == 1 + add x4, x4, #60 +.endif + + ldr d4, [x2, x10, lsl #3] + add x10, x2, x11, lsl #3 + ldr d5, [x2, x12, lsl #3] + add x12, x2, x13, lsl #3 + ldr d6, [x2, x14, lsl #3] + add x14, x2, x15, lsl #3 + ldr d7, [x2, x16, lsl #3] + add x16, x2, x17, lsl #3 + ld1 { v4.d }[1], [x10] + ld1 { v5.d }[1], [x12] + ld1 { v6.d }[1], [x14] + ld1 { v7.d }[1], [x16] +.endif +.endm + +// Single 15-point FFT (see doc/transforms.md and the AVX2 FFT15); each ymm +// becomes a pair of quads holding 2 complex each. Uses the derived constants +// and tab_15pt in v24 (dc fold sign); with hoist_strides=1 the caller +// provides x6/x7 = stride*3/*5. +.macro FFT15_CORE hoist_strides=0 +.if \hoist_strides == 0 + add x6, x3, x3, lsl #1 // stride*3 + add x7, x3, x3, lsl #2 // stride*5 +.endif + add x8, x1, x7 // &out[5] + add x9, x8, x7 // &out[10] + + // 4x parallel 3pt over in[3..14] (the in[11..14] -+ signs are folded + // into the twiddles: k = in[11..14] + Q4 -+ Q0), interleaved with the + // dc 3pt over in[0..2] ([dc] tagged, v0 = dc[0] dup, v1 = dc[1,2]) + fsub v16.4s, v2.4s, v4.4s // q[0,1]raw = in[3,4]-in[7,8] + ext v26.16b, v0.16b, v0.16b, #8 // [dc] (in1, in0) + fsub v17.4s, v3.4s, v5.4s // q[2,3]raw = in[5,6]-in[9,10] + fadd v27.4s, v0.4s, v26.4s // [dc] pc[1]raw = in0+in1 + fadd v2.4s, v2.4s, v4.4s // q[4,5]raw + fsub v20.4s, v0.4s, v26.4s // [dc] (in0-in1, in1-in0) + fadd v3.4s, v3.4s, v5.4s // q[6,7]raw + rev64 v20.4s, v20.4s // [dc] pc[0]raw in hi half + rev64 v16.4s, v16.4s // q[0,1]raw re/im-swapped + ext v21.16b, v20.16b, v27.16b, #8 // [dc] (pc[0], pc[1]) + rev64 v17.4s, v17.4s // q[2,3]raw re/im-swapped + fadd v0.4s, v1.4s, v27.4s // [dc] dc[0] = in2 + pc[1]raw (dup) + fadd v22.4s, v6.4s, v2.4s // y[0,1] = in[11,12] + q[4,5] + fmul v21.4s, v21.4s, v30.4s // [dc] pc[0,1] scaled by tab[8..11] + fadd v23.4s, v7.4s, v3.4s // y[2,3] = in[13,14] + q[6,7] + fmul v16.4s, v16.4s, v8.4s // Q0[0,1] + ext v26.16b, v21.16b, v21.16b, #8 // [dc] (pc[1], pc[0]) + fmul v17.4s, v17.4s, v8.4s // Q0[2,3] + fmla v21.4s, v26.4s, v24.4s // [dc] (dc[1]_int, dc[2]_int) + fmla v6.4s, v2.4s, v9.4s // M[0,1] = in[11,12] + q*Q4mult + fmla v7.4s, v3.4s, v9.4s // M[2,3] = in[13,14] + q*Q4mult + fmla v1.4s, v21.4s, v31.4s // [dc] v1 = (dc[1], dc[2]) — DC done + fsub v4.4s, v6.4s, v16.4s // k[0,1] = M[0,1] - Q0[0,1] + fsub v5.4s, v7.4s, v17.4s // k[2,3] = M[2,3] - Q0[2,3] + fadd v2.4s, v6.4s, v16.4s // k[4,5] = M[0,1] + Q0[0,1] + fadd v3.4s, v7.4s, v17.4s // k[6,7] = M[2,3] + Q0[2,3] + + // 4pt butterflies on y (v22,v23), k[0..3] (v4,v5), k[4..7] (v2,v3); + // one shared swapped operand per pair leaves the hi t's half-swapped, + // which the dup-symmetric twiddles absorb and the output stage uses + ext v16.16b, v23.16b, v23.16b, #8 // (y3, y2) + ext v17.16b, v5.16b, v5.16b, #8 // (k3, k2) + ext v20.16b, v3.16b, v3.16b, #8 // (k7, k6) + fsub v21.4s, v22.4s, v16.4s // (t3, t2) + fadd v22.4s, v22.4s, v16.4s // (t0, t1) + fsub v26.4s, v4.4s, v17.4s // (t7, t6) + fadd v4.4s, v4.4s, v17.4s // (t4, t5) + fsub v27.4s, v2.4s, v20.4s // (t11, t10) + fadd v2.4s, v2.4s, v20.4s // (t8, t9) + + // the 3 direct outputs: out[0,10,5] = dc[0,1,2] + t[0,4,8] + t[1,5,9] + ext v16.16b, v22.16b, v22.16b, #8 // (t1, t0) + zip1 v17.2d, v4.2d, v2.2d // (t4, t8) + zip2 v20.2d, v4.2d, v2.2d // (t5, t9) + fadd v16.4s, v16.4s, v22.4s // t[0]+t[1] + fadd v17.4s, v17.4s, v20.4s // (t[4]+t[5], t[8]+t[9]) + fadd v16.4s, v16.4s, v0.4s // out[0] + fadd v17.4s, v17.4s, v1.4s // (out[10], out[5]) + st1 { v16.d }[0], [x1] + st1 { v17.d }[1], [x8] + st1 { v17.d }[0], [x9] + + // twiddles; swap(t * tab) = swap(t) * tab as every multiplier is + // dup-symmetric. lo chunks seed the accumulator with dc[] (= the + // output stage's dc preadd): m = dc + t*v25 - swap(t)*v28; hi chunks + // r = t*v29 + swap(t)*v10, v10's +- giving r[3] += t[2]/r[2] -= t[3] + // in half-swapped order. Accumulates are spread out for the A53 + ext v16.16b, v22.16b, v22.16b, #8 // (t1, t0) + mov v6.16b, v0.16b // m0 = dc[0] + ext v17.16b, v21.16b, v21.16b, #8 // (t2, t3) + fmul v7.4s, v21.4s, v29.4s // (r3, r2) + fmla v6.4s, v22.4s, v25.4s // m0 += t[0,1]*r_lo + dup v18.2d, v1.d[0] // m1 = dc[1] + fmla v18.4s, v4.4s, v25.4s // m1 += t[4,5]*r_lo + fmls v6.4s, v16.4s, v28.4s // m0 -= swap*nt_lo + ext v16.16b, v4.16b, v4.16b, #8 // (t5, t4) + fmla v7.4s, v17.4s, v10.4s // (r3, r2) += swap*nt_hi + ext v17.16b, v26.16b, v26.16b, #8 // (t6, t7) + fmls v18.4s, v16.4s, v28.4s // m1 -= swap*nt_lo + fmul v23.4s, v26.4s, v29.4s // (r7, r6) + dup v19.2d, v1.d[1] // m2 = dc[2] + fmla v19.4s, v2.4s, v25.4s // m2 += t[8,9]*r_lo + ext v16.16b, v2.16b, v2.16b, #8 // (t9, t8) + fmla v23.4s, v17.4s, v10.4s // (r7, r6) + ext v17.16b, v27.16b, v27.16b, #8 // (t10, t11) + fmul v5.4s, v27.4s, v29.4s // (r11, r10) + fmls v19.4s, v16.4s, v28.4s // m2 -= swap*nt_lo + fmla v5.4s, v17.4s, v10.4s // (r11, r10) + + // output butterflies around rot(x) = (x.im, -x.re): out = m +- rot(r_hi). + // The half-swap makes rev(r_hi) a plain rev64, and u = rev*v31 is + // exact (+-1.0), so each +-rot pair is one non-destructive fsub/fadd + rev64 v16.4s, v7.4s // (r3.im, r3.re, r2.im, r2.re) + rev64 v17.4s, v23.4s // (r7.im, r7.re, r6.im, r6.re) + rev64 v20.4s, v5.4s // (r11.im, ..., r10.re) + fmul v16.4s, v16.4s, v31.4s // u0 + fmul v17.4s, v17.4s, v31.4s // u1 + fmul v20.4s, v20.4s, v31.4s // u2 + fsub v7.4s, v6.4s, v16.4s // (out6, out3) = m0 + rot + fadd v6.4s, v6.4s, v16.4s // (out9, out12) = m0 - rot + fsub v23.4s, v18.4s, v17.4s // (out1, out13) + fadd v22.4s, v18.4s, v17.4s // (out4, out7) + fsub v5.4s, v19.4s, v20.4s // (out11, out8) + fadd v4.4s, v19.4s, v20.4s // (out14, out2) + + add x10, x1, x6, lsl #1 // &out[6] + add x11, x1, x6 // &out[3] + st1 { v7.d }[0], [x10] + st1 { v7.d }[1], [x11] + add x12, x8, x3, lsl #2 // &out[9] + add x13, x1, x6, lsl #2 // &out[12] + st1 { v6.d }[0], [x12] + st1 { v6.d }[1], [x13] + add x10, x1, x3 // &out[1] + add x11, x9, x6 // &out[13] + st1 { v23.d }[0], [x10] + st1 { v23.d }[1], [x11] + add x12, x1, x3, lsl #2 // &out[4] + add x13, x8, x3, lsl #1 // &out[7] + st1 { v22.d }[0], [x12] + st1 { v22.d }[1], [x13] + add x10, x9, x3 // &out[11] + add x11, x1, x3, lsl #3 // &out[8] + st1 { v5.d }[0], [x10] + st1 { v5.d }[1], [x11] + add x12, x9, x3, lsl #2 // &out[14] + add x13, x1, x3, lsl #1 // &out[2] + st1 { v4.d }[0], [x12] + st1 { v4.d }[1], [x13] +.endm + +.macro FFT15_FN name, no_perm +function ff_tx_fft15_\name\()_neon, export=1 + stp d8, d9, [sp, #-32]! + str d10, [sp, #16] + + SETUP_LUT \no_perm + + movrel x5, X(ff_tx_tab_53_float) + ld1 { v28.4s, v29.4s, v30.4s }, [x5] // 5pt cos, 5pt sin, 3pt + + movrel x5, tab_15pt + ld1 { v24.4s }, [x5] // sign mask + + LOAD_SUBADD + FFT15_DERIVE_CONSTS + + FFT15_LOAD \no_perm + + FFT15_CORE + + ldr d10, [sp, #16] + ldp d8, d9, [sp], #32 + ret +endfunc +.endm + +FFT15_FN float, 0 +FFT15_FN ns_float, 1 + .macro SETUP_SR_RECOMB len, re, im, dec ldr w5, =(\len - 4*7) movrel \re, X(ff_tx_tab_\len\()_float) -- To stop receiving notification emails like this one, please contact [email protected]. _______________________________________________ ffmpeg-cvslog mailing list -- [email protected] To unsubscribe send an email to [email protected]
