Commit 357bb4965f for ffmpeg
commit 357bb4965f276294ef051f06e7ebebc813e46f64
Author: Lynne <dev@lynne.ee>
Date: Thu Jun 4 15:04:13 2026 +0900
lavu/tx: add AArch64 NEON fft_pfa_15xM
Same as the x86 version.
diff --git a/libavutil/aarch64/tx_float_init.c b/libavutil/aarch64/tx_float_init.c
index 47f1e12700..a049562609 100644
--- a/libavutil/aarch64/tx_float_init.c
+++ b/libavutil/aarch64/tx_float_init.c
@@ -19,6 +19,7 @@
#define TX_FLOAT
#include "libavutil/tx_priv.h"
#include "libavutil/attributes.h"
+#include "libavutil/mem.h"
#include "libavutil/aarch64/cpu.h"
TX_DECL_FN(fft2, neon)
@@ -34,6 +35,8 @@ TX_DECL_FN(fft32, neon)
TX_DECL_FN(fft32_ns, neon)
TX_DECL_FN(fft_sr, neon)
TX_DECL_FN(fft_sr_ns, neon)
+TX_DECL_FN(fft_pfa_15xM, neon)
+TX_DECL_FN(fft_pfa_15xM_ns, neon)
static av_cold int neon_init(AVTXContext *s, const FFTXCodelet *cd,
uint64_t flags, FFTXCodeletOptions *opts,
@@ -46,11 +49,29 @@ static av_cold int neon_init(AVTXContext *s, const FFTXCodelet *cd,
return ff_tx_gen_split_radix_parity_revtab(s, len, inv, opts, 8, 0);
}
+/* Reorder one 15-point map so the loads in the pre-permuted assembly path
+ * become simple contiguous chunks. Mirrors the x86 FFT15 init. */
+static void fft15_permute_map(int *map)
+{
+ int cnt = 0, tmp[15];
+ memcpy(tmp, map, 15*sizeof(*tmp));
+ for (int i = 1; i < 15; i += 3)
+ map[cnt++] = tmp[i];
+ for (int i = 2; i < 15; i += 3)
+ map[cnt++] = tmp[i];
+ for (int i = 0; i < 15; i += 3)
+ map[cnt++] = tmp[i];
+ memmove(&map[7], &map[6], 4*sizeof(int));
+ memmove(&map[3], &map[1], 4*sizeof(int));
+ map[1] = tmp[2];
+ map[2] = tmp[0];
+}
+
static av_cold int fft15_init(AVTXContext *s, const FFTXCodelet *cd,
uint64_t flags, FFTXCodeletOptions *opts,
int len, int inv, const void *scale)
{
- int ret, cnt = 0, tmp[15];
+ int ret;
FFTXCodeletOptions sub_opts = { .map_dir = FF_TX_MAP_GATHER };
ff_tx_init_tabs_float(len);
@@ -58,19 +79,44 @@ static av_cold int fft15_init(AVTXContext *s, const FFTXCodelet *cd,
if ((ret = ff_tx_gen_pfa_input_map(s, &sub_opts, 3, 5)) < 0)
return ret;
- /* Reorder the 15-pt map so the loads in the pre-permuted assembly path
- * become simple contiguous chunks. Mirrors the x86 FFT15 init. */
- memcpy(tmp, s->map, 15*sizeof(*tmp));
- for (int i = 1; i < 15; i += 3)
- s->map[cnt++] = tmp[i];
- for (int i = 2; i < 15; i += 3)
- s->map[cnt++] = tmp[i];
- for (int i = 0; i < 15; i += 3)
- s->map[cnt++] = tmp[i];
- memmove(&s->map[7], &s->map[6], 4*sizeof(int));
- memmove(&s->map[3], &s->map[1], 4*sizeof(int));
- s->map[1] = tmp[2];
- s->map[2] = tmp[0];
+ fft15_permute_map(s->map);
+
+ return 0;
+}
+
+/* 15xM prime-factor FFT: M inlined 15-point transforms followed by 15 calls to
+ * a power-of-two subtransform. Mirrors the x86 fft_pfa_15xM, but the aarch64
+ * ABI lets us call the subtransform normally, so no FF_TX_ASM_CALL is needed. */
+static av_cold int fft_pfa_init(AVTXContext *s, const FFTXCodelet *cd,
+ uint64_t flags, FFTXCodeletOptions *opts,
+ int len, int inv, const void *scale)
+{
+ int ret;
+ int sub_len = len / cd->factors[0];
+ FFTXCodeletOptions sub_opts = { .map_dir = FF_TX_MAP_SCATTER };
+
+ flags &= ~FF_TX_OUT_OF_PLACE; /* We want the subtransform to be */
+ flags |= AV_TX_INPLACE; /* in-place */
+ flags |= FF_TX_PRESHUFFLE; /* This function handles the permute step */
+
+ if ((ret = ff_tx_init_subtx(s, TX_TYPE(FFT), flags, &sub_opts,
+ sub_len, inv, scale)))
+ return ret;
+
+ if ((ret = ff_tx_gen_compound_mapping(s, opts, s->inv,
+ cd->factors[0], sub_len)))
+ return ret;
+
+ /* The 15-point transform is itself a compound one, so embed its input map
+ * and apply the same load-friendly reorder used by fft15_init. */
+ TX_EMBED_INPUT_PFA_MAP(s->map, len, 3, 5);
+ for (int k = 0; k < sub_len; k++)
+ fft15_permute_map(&s->map[k*15]);
+
+ if (!(s->tmp = av_malloc(len*sizeof(*s->tmp))))
+ return AVERROR(ENOMEM);
+
+ ff_tx_init_tabs_float(len / sub_len);
return 0;
}
@@ -93,5 +139,8 @@ const FFTXCodelet * const ff_tx_codelet_list_float_aarch64[] = {
TX_DEF(fft_sr, FFT, 64, 131072, 2, 0, 128, neon_init, neon, NEON, 0, 0),
TX_DEF(fft_sr_ns, FFT, 64, 131072, 2, 0, 192, neon_init, neon, NEON, AV_TX_INPLACE | FF_TX_PRESHUFFLE, 0),
+ TX_DEF(fft_pfa_15xM, FFT, 60, TX_LEN_UNLIMITED, 15, 2, 128, fft_pfa_init, neon, NEON, AV_TX_INPLACE, 0),
+ TX_DEF(fft_pfa_15xM_ns, FFT, 60, TX_LEN_UNLIMITED, 15, 2, 192, fft_pfa_init, neon, NEON, AV_TX_INPLACE | FF_TX_PRESHUFFLE, 0),
+
NULL,
};
diff --git a/libavutil/aarch64/tx_float_neon.S b/libavutil/aarch64/tx_float_neon.S
index 255dc3acaa..2db0e95740 100644
--- a/libavutil/aarch64/tx_float_neon.S
+++ b/libavutil/aarch64/tx_float_neon.S
@@ -671,6 +671,105 @@ endfunc
FFT15_FN float, 0
FFT15_FN ns_float, 1
+// 15xM PFA (len = 15*M, M a power of two), like the x86 fft_pfa_15xM:
+// dim1 = M 15pt transforms scattered into s->tmp at sub_map[i], spaced M
+// apart; dim2 = 15 in-place M-pt subtransforms (plain blr, the uniform ABI
+// needs no asm-call variant); post = out[i*stride] = s->tmp[out_map[i]].
+// AVTXContext offsets: len=0, map=8, tmp=24, sub=32, fn[0]=40.
+.macro PFA_15_FN name, no_perm
+function ff_tx_fft_pfa_15xM_\name\()_neon, export=1
+ AARCH64_SIGN_LINK_REGISTER
+ stp x29, x30, [sp, #-128]!
+ mov x29, sp
+ stp x19, x20, [sp, #16]
+ stp x21, x22, [sp, #32]
+ stp x23, x24, [sp, #48]
+ stp x25, x26, [sp, #64]
+ stp x27, x28, [sp, #80]
+ stp d8, d9, [sp, #96]
+ str d10, [sp, #112]
+
+ mov x25, x0 // root context
+ mov x26, x1 // user out
+ mov x27, x3 // user stride (bytes)
+
+ ldr w19, [x0, #0] // len = 15*M
+ ldr x20, [x0, #24] // s->tmp
+ ldr x23, [x0, #32] // s->sub
+ ldr w24, [x23, #0] // M
+ ldr x28, [x23, #8] // sub_map
+ lsl x3, x24, #3 // M*8 = 15pt output stride
+.if \no_perm == 0
+ ldr x4, [x0, #8] // in_map, advanced by the loads
+.endif
+
+ movrel x5, X(ff_tx_tab_53_float)
+ ld1 { v28.4s, v29.4s, v30.4s }, [x5]
+ movrel x5, tab_15pt
+ ld1 { v24.4s }, [x5]
+ LOAD_SUBADD
+ FFT15_DERIVE_CONSTS
+ add x6, x3, x3, lsl #1 // stride*3
+ add x7, x3, x3, lsl #2 // stride*5
+1:
+ ldr w10, [x28], #4 // sub_map[i]
+ add x1, x20, x10, lsl #3
+ FFT15_LOAD \no_perm, advance=1
+ FFT15_CORE hoist_strides=1
+ subs w19, w19, #15
+ b.gt 1b
+
+ ldr x28, [x25, #40] // ctx->fn[0]
+ mov x21, x20 // column base
+ mov w19, #15
+2:
+ mov x0, x23
+ mov x1, x21
+ mov x2, x21
+ mov x3, #8
+ blr x28 // M-point FFT, in-place
+ add x21, x21, x24, lsl #3
+ subs w19, w19, #1
+ b.gt 2b
+
+ // out[i*stride] = s->tmp[out_map[i]], unrolled by 4 (len % 60 == 0)
+ ldr w19, [x25, #0] // len
+ ldr x22, [x25, #8] // s->map
+ add x22, x22, x19, lsl #2 // out_map = map + len
+ mov x10, x26
+ lsl x11, x27, #1 // 2*stride
+ add x12, x27, x11 // 3*stride
+3:
+ ldp w13, w14, [x22], #8 // out_map[i, i+1]
+ ldp w15, w16, [x22], #8 // out_map[i+2, i+3]
+ ldr d0, [x20, x13, lsl #3]
+ ldr d1, [x20, x14, lsl #3]
+ ldr d2, [x20, x15, lsl #3]
+ ldr d3, [x20, x16, lsl #3]
+ str d0, [x10]
+ str d1, [x10, x27]
+ str d2, [x10, x11]
+ str d3, [x10, x12]
+ add x10, x10, x11, lsl #1
+ subs w19, w19, #4
+ b.gt 3b
+
+ ldp x19, x20, [sp, #16]
+ ldp x21, x22, [sp, #32]
+ ldp x23, x24, [sp, #48]
+ ldp x25, x26, [sp, #64]
+ ldp x27, x28, [sp, #80]
+ ldp d8, d9, [sp, #96]
+ ldr d10, [sp, #112]
+ ldp x29, x30, [sp], #128
+ AARCH64_VALIDATE_LINK_REGISTER
+ ret
+endfunc
+.endm
+
+PFA_15_FN float, 0
+PFA_15_FN ns_float, 1
+
.macro SETUP_SR_RECOMB len, re, im, dec
ldr w5, =(\len - 4*7)
movrel \re, X(ff_tx_tab_\len\()_float)