Commit 31f5c05c4e for ffmpeg
commit 31f5c05c4e0a1422eff9c87358c25488e0104278
Author: Lynne <dev@lynne.ee>
Date: Thu Jun 4 06:03:55 2026 +0900
lavu/tx: add AArch64 NEON fft15 codelet
Based on the C code in doc/transforms.md.
diff --git a/libavutil/aarch64/tx_float_init.c b/libavutil/aarch64/tx_float_init.c
index 8300472c4c..47f1e12700 100644
--- a/libavutil/aarch64/tx_float_init.c
+++ b/libavutil/aarch64/tx_float_init.c
@@ -26,6 +26,8 @@ TX_DECL_FN(fft4_fwd, neon)
TX_DECL_FN(fft4_inv, neon)
TX_DECL_FN(fft8, neon)
TX_DECL_FN(fft8_ns, neon)
+TX_DECL_FN(fft15, neon)
+TX_DECL_FN(fft15_ns, neon)
TX_DECL_FN(fft16, neon)
TX_DECL_FN(fft16_ns, neon)
TX_DECL_FN(fft32, neon)
@@ -44,6 +46,35 @@ static av_cold int neon_init(AVTXContext *s, const FFTXCodelet *cd,
return ff_tx_gen_split_radix_parity_revtab(s, len, inv, opts, 8, 0);
}
+static av_cold int fft15_init(AVTXContext *s, const FFTXCodelet *cd,
+ uint64_t flags, FFTXCodeletOptions *opts,
+ int len, int inv, const void *scale)
+{
+ int ret, cnt = 0, tmp[15];
+ FFTXCodeletOptions sub_opts = { .map_dir = FF_TX_MAP_GATHER };
+
+ ff_tx_init_tabs_float(len);
+
+ if ((ret = ff_tx_gen_pfa_input_map(s, &sub_opts, 3, 5)) < 0)
+ return ret;
+
+ /* Reorder the 15-pt map so the loads in the pre-permuted assembly path
+ * become simple contiguous chunks. Mirrors the x86 FFT15 init. */
+ memcpy(tmp, s->map, 15*sizeof(*tmp));
+ for (int i = 1; i < 15; i += 3)
+ s->map[cnt++] = tmp[i];
+ for (int i = 2; i < 15; i += 3)
+ s->map[cnt++] = tmp[i];
+ for (int i = 0; i < 15; i += 3)
+ s->map[cnt++] = tmp[i];
+ memmove(&s->map[7], &s->map[6], 4*sizeof(int));
+ memmove(&s->map[3], &s->map[1], 4*sizeof(int));
+ s->map[1] = tmp[2];
+ s->map[2] = tmp[0];
+
+ return 0;
+}
+
const FFTXCodelet * const ff_tx_codelet_list_float_aarch64[] = {
TX_DEF(fft2, FFT, 2, 2, 2, 0, 128, NULL, neon, NEON, AV_TX_INPLACE, 0),
TX_DEF(fft2, FFT, 2, 2, 2, 0, 192, neon_init, neon, NEON, AV_TX_INPLACE | FF_TX_PRESHUFFLE, 0),
@@ -52,6 +83,8 @@ const FFTXCodelet * const ff_tx_codelet_list_float_aarch64[] = {
TX_DEF(fft4_inv, FFT, 4, 4, 2, 0, 128, NULL, neon, NEON, AV_TX_INPLACE | FF_TX_INVERSE_ONLY, 0),
TX_DEF(fft8, FFT, 8, 8, 2, 0, 128, neon_init, neon, NEON, AV_TX_INPLACE, 0),
TX_DEF(fft8_ns, FFT, 8, 8, 2, 0, 192, neon_init, neon, NEON, AV_TX_INPLACE | FF_TX_PRESHUFFLE, 0),
+ TX_DEF(fft15, FFT, 15, 15, 15, 0, 128, fft15_init, neon, NEON, AV_TX_INPLACE, 0),
+ TX_DEF(fft15_ns, FFT, 15, 15, 15, 0, 192, fft15_init, neon, NEON, AV_TX_INPLACE | FF_TX_PRESHUFFLE, 0),
TX_DEF(fft16, FFT, 16, 16, 2, 0, 128, neon_init, neon, NEON, AV_TX_INPLACE, 0),
TX_DEF(fft16_ns, FFT, 16, 16, 2, 0, 192, neon_init, neon, NEON, AV_TX_INPLACE | FF_TX_PRESHUFFLE, 0),
TX_DEF(fft32, FFT, 32, 32, 2, 0, 128, neon_init, neon, NEON, AV_TX_INPLACE, 0),
diff --git a/libavutil/aarch64/tx_float_neon.S b/libavutil/aarch64/tx_float_neon.S
index 12c4e880dc..255dc3acaa 100644
--- a/libavutil/aarch64/tx_float_neon.S
+++ b/libavutil/aarch64/tx_float_neon.S
@@ -438,6 +438,239 @@ endfunc
FFT16_FN float, 0
FFT16_FN ns_float, 1
+const tab_15pt, align=4
+ .float 1.0, 1.0, -1.0, -1.0
+endconst
+
+// Tab_53 twiddles (v28..v30) duplicated/pre-signed once, instead of per
+// transform: v8/v9 = -+tab[8,9]/[10,11], v25/v28/v29 = tab[0,1]/[2,3]/[4,5],
+// v10 = +-tab[6,7]. v30 keeps tab[8..11]. Callers preserve d8-d10.
+.macro FFT15_DERIVE_CONSTS
+ dup v8.2d, v30.d[0]
+ dup v9.2d, v30.d[1]
+ dup v10.2d, v29.d[1]
+ dup v25.2d, v28.d[0]
+ dup v28.2d, v28.d[1]
+ dup v29.2d, v29.d[0]
+ fmul v8.4s, v8.4s, v31.4s
+ fmul v9.4s, v9.4s, v31.4s
+ fmul v10.4s, v10.4s, v24.4s
+.endm
+
+.macro FFT15_LOAD no_perm, advance=0
+.if \no_perm == 1
+ // Writebacks leave x2 a whole transform (120B) ahead for the PFA loop
+ ld1 { v0.4s }, [x2], #16 // in[0,1]
+ ld1r { v1.2d }, [x2], #8 // in[2] duplicated
+ ld1 { v2.4s, v3.4s, v4.4s }, [x2], #48 // in[3..8]
+ ld1 { v5.4s, v6.4s, v7.4s }, [x2], #48 // in[9..14]
+.else
+ ldp w10, w11, [x4] // lut[0,1]
+ ldr w12, [x4, #8] // lut[2]
+ ldp w13, w14, [x4, #12] // lut[3,4]
+ ldp w15, w16, [x4, #20] // lut[5,6]
+
+ ldr d0, [x2, x10, lsl #3]
+ add x10, x2, x11, lsl #3
+ add x12, x2, x12, lsl #3
+ ld1 { v0.d }[1], [x10]
+ ld1r { v1.2d }, [x12]
+
+ ldr d2, [x2, x13, lsl #3]
+ add x13, x2, x14, lsl #3
+ ldr d3, [x2, x15, lsl #3]
+ add x15, x2, x16, lsl #3
+ ld1 { v2.d }[1], [x13]
+ ld1 { v3.d }[1], [x15]
+
+ ldp w10, w11, [x4, #28] // lut[7,8]
+ ldp w12, w13, [x4, #36] // lut[9,10]
+ ldp w14, w15, [x4, #44] // lut[11,12]
+ ldp w16, w17, [x4, #52] // lut[13,14]
+.if \advance == 1
+ add x4, x4, #60
+.endif
+
+ ldr d4, [x2, x10, lsl #3]
+ add x10, x2, x11, lsl #3
+ ldr d5, [x2, x12, lsl #3]
+ add x12, x2, x13, lsl #3
+ ldr d6, [x2, x14, lsl #3]
+ add x14, x2, x15, lsl #3
+ ldr d7, [x2, x16, lsl #3]
+ add x16, x2, x17, lsl #3
+ ld1 { v4.d }[1], [x10]
+ ld1 { v5.d }[1], [x12]
+ ld1 { v6.d }[1], [x14]
+ ld1 { v7.d }[1], [x16]
+.endif
+.endm
+
+// Single 15-point FFT (see doc/transforms.md and the AVX2 FFT15); each ymm
+// becomes a pair of quads holding 2 complex each. Uses the derived constants
+// and tab_15pt in v24 (dc fold sign); with hoist_strides=1 the caller
+// provides x6/x7 = stride*3/*5.
+.macro FFT15_CORE hoist_strides=0
+.if \hoist_strides == 0
+ add x6, x3, x3, lsl #1 // stride*3
+ add x7, x3, x3, lsl #2 // stride*5
+.endif
+ add x8, x1, x7 // &out[5]
+ add x9, x8, x7 // &out[10]
+
+ // 4x parallel 3pt over in[3..14] (the in[11..14] -+ signs are folded
+ // into the twiddles: k = in[11..14] + Q4 -+ Q0), interleaved with the
+ // dc 3pt over in[0..2] ([dc] tagged, v0 = dc[0] dup, v1 = dc[1,2])
+ fsub v16.4s, v2.4s, v4.4s // q[0,1]raw = in[3,4]-in[7,8]
+ ext v26.16b, v0.16b, v0.16b, #8 // [dc] (in1, in0)
+ fsub v17.4s, v3.4s, v5.4s // q[2,3]raw = in[5,6]-in[9,10]
+ fadd v27.4s, v0.4s, v26.4s // [dc] pc[1]raw = in0+in1
+ fadd v2.4s, v2.4s, v4.4s // q[4,5]raw
+ fsub v20.4s, v0.4s, v26.4s // [dc] (in0-in1, in1-in0)
+ fadd v3.4s, v3.4s, v5.4s // q[6,7]raw
+ rev64 v20.4s, v20.4s // [dc] pc[0]raw in hi half
+ rev64 v16.4s, v16.4s // q[0,1]raw re/im-swapped
+ ext v21.16b, v20.16b, v27.16b, #8 // [dc] (pc[0], pc[1])
+ rev64 v17.4s, v17.4s // q[2,3]raw re/im-swapped
+ fadd v0.4s, v1.4s, v27.4s // [dc] dc[0] = in2 + pc[1]raw (dup)
+ fadd v22.4s, v6.4s, v2.4s // y[0,1] = in[11,12] + q[4,5]
+ fmul v21.4s, v21.4s, v30.4s // [dc] pc[0,1] scaled by tab[8..11]
+ fadd v23.4s, v7.4s, v3.4s // y[2,3] = in[13,14] + q[6,7]
+ fmul v16.4s, v16.4s, v8.4s // Q0[0,1]
+ ext v26.16b, v21.16b, v21.16b, #8 // [dc] (pc[1], pc[0])
+ fmul v17.4s, v17.4s, v8.4s // Q0[2,3]
+ fmla v21.4s, v26.4s, v24.4s // [dc] (dc[1]_int, dc[2]_int)
+ fmla v6.4s, v2.4s, v9.4s // M[0,1] = in[11,12] + q*Q4mult
+ fmla v7.4s, v3.4s, v9.4s // M[2,3] = in[13,14] + q*Q4mult
+ fmla v1.4s, v21.4s, v31.4s // [dc] v1 = (dc[1], dc[2]) — DC done
+ fsub v4.4s, v6.4s, v16.4s // k[0,1] = M[0,1] - Q0[0,1]
+ fsub v5.4s, v7.4s, v17.4s // k[2,3] = M[2,3] - Q0[2,3]
+ fadd v2.4s, v6.4s, v16.4s // k[4,5] = M[0,1] + Q0[0,1]
+ fadd v3.4s, v7.4s, v17.4s // k[6,7] = M[2,3] + Q0[2,3]
+
+ // 4pt butterflies on y (v22,v23), k[0..3] (v4,v5), k[4..7] (v2,v3);
+ // one shared swapped operand per pair leaves the hi t's half-swapped,
+ // which the dup-symmetric twiddles absorb and the output stage uses
+ ext v16.16b, v23.16b, v23.16b, #8 // (y3, y2)
+ ext v17.16b, v5.16b, v5.16b, #8 // (k3, k2)
+ ext v20.16b, v3.16b, v3.16b, #8 // (k7, k6)
+ fsub v21.4s, v22.4s, v16.4s // (t3, t2)
+ fadd v22.4s, v22.4s, v16.4s // (t0, t1)
+ fsub v26.4s, v4.4s, v17.4s // (t7, t6)
+ fadd v4.4s, v4.4s, v17.4s // (t4, t5)
+ fsub v27.4s, v2.4s, v20.4s // (t11, t10)
+ fadd v2.4s, v2.4s, v20.4s // (t8, t9)
+
+ // the 3 direct outputs: out[0,10,5] = dc[0,1,2] + t[0,4,8] + t[1,5,9]
+ ext v16.16b, v22.16b, v22.16b, #8 // (t1, t0)
+ zip1 v17.2d, v4.2d, v2.2d // (t4, t8)
+ zip2 v20.2d, v4.2d, v2.2d // (t5, t9)
+ fadd v16.4s, v16.4s, v22.4s // t[0]+t[1]
+ fadd v17.4s, v17.4s, v20.4s // (t[4]+t[5], t[8]+t[9])
+ fadd v16.4s, v16.4s, v0.4s // out[0]
+ fadd v17.4s, v17.4s, v1.4s // (out[10], out[5])
+ st1 { v16.d }[0], [x1]
+ st1 { v17.d }[1], [x8]
+ st1 { v17.d }[0], [x9]
+
+ // twiddles; swap(t * tab) = swap(t) * tab as every multiplier is
+ // dup-symmetric. lo chunks seed the accumulator with dc[] (= the
+ // output stage's dc preadd): m = dc + t*v25 - swap(t)*v28; hi chunks
+ // r = t*v29 + swap(t)*v10, v10's +- giving r[3] += t[2]/r[2] -= t[3]
+ // in half-swapped order. Accumulates are spread out for the A53
+ ext v16.16b, v22.16b, v22.16b, #8 // (t1, t0)
+ mov v6.16b, v0.16b // m0 = dc[0]
+ ext v17.16b, v21.16b, v21.16b, #8 // (t2, t3)
+ fmul v7.4s, v21.4s, v29.4s // (r3, r2)
+ fmla v6.4s, v22.4s, v25.4s // m0 += t[0,1]*r_lo
+ dup v18.2d, v1.d[0] // m1 = dc[1]
+ fmla v18.4s, v4.4s, v25.4s // m1 += t[4,5]*r_lo
+ fmls v6.4s, v16.4s, v28.4s // m0 -= swap*nt_lo
+ ext v16.16b, v4.16b, v4.16b, #8 // (t5, t4)
+ fmla v7.4s, v17.4s, v10.4s // (r3, r2) += swap*nt_hi
+ ext v17.16b, v26.16b, v26.16b, #8 // (t6, t7)
+ fmls v18.4s, v16.4s, v28.4s // m1 -= swap*nt_lo
+ fmul v23.4s, v26.4s, v29.4s // (r7, r6)
+ dup v19.2d, v1.d[1] // m2 = dc[2]
+ fmla v19.4s, v2.4s, v25.4s // m2 += t[8,9]*r_lo
+ ext v16.16b, v2.16b, v2.16b, #8 // (t9, t8)
+ fmla v23.4s, v17.4s, v10.4s // (r7, r6)
+ ext v17.16b, v27.16b, v27.16b, #8 // (t10, t11)
+ fmul v5.4s, v27.4s, v29.4s // (r11, r10)
+ fmls v19.4s, v16.4s, v28.4s // m2 -= swap*nt_lo
+ fmla v5.4s, v17.4s, v10.4s // (r11, r10)
+
+ // output butterflies around rot(x) = (x.im, -x.re): out = m +- rot(r_hi).
+ // The half-swap makes rev(r_hi) a plain rev64, and u = rev*v31 is
+ // exact (+-1.0), so each +-rot pair is one non-destructive fsub/fadd
+ rev64 v16.4s, v7.4s // (r3.im, r3.re, r2.im, r2.re)
+ rev64 v17.4s, v23.4s // (r7.im, r7.re, r6.im, r6.re)
+ rev64 v20.4s, v5.4s // (r11.im, ..., r10.re)
+ fmul v16.4s, v16.4s, v31.4s // u0
+ fmul v17.4s, v17.4s, v31.4s // u1
+ fmul v20.4s, v20.4s, v31.4s // u2
+ fsub v7.4s, v6.4s, v16.4s // (out6, out3) = m0 + rot
+ fadd v6.4s, v6.4s, v16.4s // (out9, out12) = m0 - rot
+ fsub v23.4s, v18.4s, v17.4s // (out1, out13)
+ fadd v22.4s, v18.4s, v17.4s // (out4, out7)
+ fsub v5.4s, v19.4s, v20.4s // (out11, out8)
+ fadd v4.4s, v19.4s, v20.4s // (out14, out2)
+
+ add x10, x1, x6, lsl #1 // &out[6]
+ add x11, x1, x6 // &out[3]
+ st1 { v7.d }[0], [x10]
+ st1 { v7.d }[1], [x11]
+ add x12, x8, x3, lsl #2 // &out[9]
+ add x13, x1, x6, lsl #2 // &out[12]
+ st1 { v6.d }[0], [x12]
+ st1 { v6.d }[1], [x13]
+ add x10, x1, x3 // &out[1]
+ add x11, x9, x6 // &out[13]
+ st1 { v23.d }[0], [x10]
+ st1 { v23.d }[1], [x11]
+ add x12, x1, x3, lsl #2 // &out[4]
+ add x13, x8, x3, lsl #1 // &out[7]
+ st1 { v22.d }[0], [x12]
+ st1 { v22.d }[1], [x13]
+ add x10, x9, x3 // &out[11]
+ add x11, x1, x3, lsl #3 // &out[8]
+ st1 { v5.d }[0], [x10]
+ st1 { v5.d }[1], [x11]
+ add x12, x9, x3, lsl #2 // &out[14]
+ add x13, x1, x3, lsl #1 // &out[2]
+ st1 { v4.d }[0], [x12]
+ st1 { v4.d }[1], [x13]
+.endm
+
+.macro FFT15_FN name, no_perm
+function ff_tx_fft15_\name\()_neon, export=1
+ stp d8, d9, [sp, #-32]!
+ str d10, [sp, #16]
+
+ SETUP_LUT \no_perm
+
+ movrel x5, X(ff_tx_tab_53_float)
+ ld1 { v28.4s, v29.4s, v30.4s }, [x5] // 5pt cos, 5pt sin, 3pt
+
+ movrel x5, tab_15pt
+ ld1 { v24.4s }, [x5] // sign mask
+
+ LOAD_SUBADD
+ FFT15_DERIVE_CONSTS
+
+ FFT15_LOAD \no_perm
+
+ FFT15_CORE
+
+ ldr d10, [sp, #16]
+ ldp d8, d9, [sp], #32
+ ret
+endfunc
+.endm
+
+FFT15_FN float, 0
+FFT15_FN ns_float, 1
+
.macro SETUP_SR_RECOMB len, re, im, dec
ldr w5, =(\len - 4*7)
movrel \re, X(ff_tx_tab_\len\()_float)