Commit 357bb4965f for ffmpeg

commit 357bb4965f276294ef051f06e7ebebc813e46f64
Author: Lynne <dev@lynne.ee>
Date:   Thu Jun 4 15:04:13 2026 +0900

    lavu/tx: add AArch64 NEON fft_pfa_15xM

    Same as the x86 version.

diff --git a/libavutil/aarch64/tx_float_init.c b/libavutil/aarch64/tx_float_init.c
index 47f1e12700..a049562609 100644
--- a/libavutil/aarch64/tx_float_init.c
+++ b/libavutil/aarch64/tx_float_init.c
@@ -19,6 +19,7 @@
 #define TX_FLOAT
 #include "libavutil/tx_priv.h"
 #include "libavutil/attributes.h"
+#include "libavutil/mem.h"
 #include "libavutil/aarch64/cpu.h"

 TX_DECL_FN(fft2,      neon)
@@ -34,6 +35,8 @@ TX_DECL_FN(fft32,     neon)
 TX_DECL_FN(fft32_ns,  neon)
 TX_DECL_FN(fft_sr,    neon)
 TX_DECL_FN(fft_sr_ns, neon)
+TX_DECL_FN(fft_pfa_15xM, neon)
+TX_DECL_FN(fft_pfa_15xM_ns, neon)

 static av_cold int neon_init(AVTXContext *s, const FFTXCodelet *cd,
                              uint64_t flags, FFTXCodeletOptions *opts,
@@ -46,11 +49,29 @@ static av_cold int neon_init(AVTXContext *s, const FFTXCodelet *cd,
         return ff_tx_gen_split_radix_parity_revtab(s, len, inv, opts, 8, 0);
 }

+/* Reorder one 15-point map so the loads in the pre-permuted assembly path
+ * become simple contiguous chunks. Mirrors the x86 FFT15 init. */
+static void fft15_permute_map(int *map)
+{
+    int cnt = 0, tmp[15];
+    memcpy(tmp, map, 15*sizeof(*tmp));
+    for (int i = 1; i < 15; i += 3)
+        map[cnt++] = tmp[i];
+    for (int i = 2; i < 15; i += 3)
+        map[cnt++] = tmp[i];
+    for (int i = 0; i < 15; i += 3)
+        map[cnt++] = tmp[i];
+    memmove(&map[7], &map[6], 4*sizeof(int));
+    memmove(&map[3], &map[1], 4*sizeof(int));
+    map[1] = tmp[2];
+    map[2] = tmp[0];
+}
+
 static av_cold int fft15_init(AVTXContext *s, const FFTXCodelet *cd,
                               uint64_t flags, FFTXCodeletOptions *opts,
                               int len, int inv, const void *scale)
 {
-    int ret, cnt = 0, tmp[15];
+    int ret;
     FFTXCodeletOptions sub_opts = { .map_dir = FF_TX_MAP_GATHER };

     ff_tx_init_tabs_float(len);
@@ -58,19 +79,44 @@ static av_cold int fft15_init(AVTXContext *s, const FFTXCodelet *cd,
     if ((ret = ff_tx_gen_pfa_input_map(s, &sub_opts, 3, 5)) < 0)
         return ret;

-    /* Reorder the 15-pt map so the loads in the pre-permuted assembly path
-     * become simple contiguous chunks. Mirrors the x86 FFT15 init. */
-    memcpy(tmp, s->map, 15*sizeof(*tmp));
-    for (int i = 1; i < 15; i += 3)
-        s->map[cnt++] = tmp[i];
-    for (int i = 2; i < 15; i += 3)
-        s->map[cnt++] = tmp[i];
-    for (int i = 0; i < 15; i += 3)
-        s->map[cnt++] = tmp[i];
-    memmove(&s->map[7], &s->map[6], 4*sizeof(int));
-    memmove(&s->map[3], &s->map[1], 4*sizeof(int));
-    s->map[1] = tmp[2];
-    s->map[2] = tmp[0];
+    fft15_permute_map(s->map);
+
+    return 0;
+}
+
+/* 15xM prime-factor FFT: M inlined 15-point transforms followed by 15 calls to
+ * a power-of-two subtransform. Mirrors the x86 fft_pfa_15xM, but the aarch64
+ * ABI lets us call the subtransform normally, so no FF_TX_ASM_CALL is needed. */
+static av_cold int fft_pfa_init(AVTXContext *s, const FFTXCodelet *cd,
+                                uint64_t flags, FFTXCodeletOptions *opts,
+                                int len, int inv, const void *scale)
+{
+    int ret;
+    int sub_len = len / cd->factors[0];
+    FFTXCodeletOptions sub_opts = { .map_dir = FF_TX_MAP_SCATTER };
+
+    flags &= ~FF_TX_OUT_OF_PLACE; /* We want the subtransform to be */
+    flags |=  AV_TX_INPLACE;      /* in-place */
+    flags |=  FF_TX_PRESHUFFLE;   /* This function handles the permute step */
+
+    if ((ret = ff_tx_init_subtx(s, TX_TYPE(FFT), flags, &sub_opts,
+                                sub_len, inv, scale)))
+        return ret;
+
+    if ((ret = ff_tx_gen_compound_mapping(s, opts, s->inv,
+                                          cd->factors[0], sub_len)))
+        return ret;
+
+    /* The 15-point transform is itself a compound one, so embed its input map
+     * and apply the same load-friendly reorder used by fft15_init. */
+    TX_EMBED_INPUT_PFA_MAP(s->map, len, 3, 5);
+    for (int k = 0; k < sub_len; k++)
+        fft15_permute_map(&s->map[k*15]);
+
+    if (!(s->tmp = av_malloc(len*sizeof(*s->tmp))))
+        return AVERROR(ENOMEM);
+
+    ff_tx_init_tabs_float(len / sub_len);

     return 0;
 }
@@ -93,5 +139,8 @@ const FFTXCodelet * const ff_tx_codelet_list_float_aarch64[] = {
     TX_DEF(fft_sr,    FFT, 64, 131072, 2, 0, 128, neon_init, neon, NEON, 0, 0),
     TX_DEF(fft_sr_ns, FFT, 64, 131072, 2, 0, 192, neon_init, neon, NEON, AV_TX_INPLACE | FF_TX_PRESHUFFLE, 0),

+    TX_DEF(fft_pfa_15xM,    FFT, 60, TX_LEN_UNLIMITED, 15, 2, 128, fft_pfa_init, neon, NEON, AV_TX_INPLACE, 0),
+    TX_DEF(fft_pfa_15xM_ns, FFT, 60, TX_LEN_UNLIMITED, 15, 2, 192, fft_pfa_init, neon, NEON, AV_TX_INPLACE | FF_TX_PRESHUFFLE, 0),
+
     NULL,
 };
diff --git a/libavutil/aarch64/tx_float_neon.S b/libavutil/aarch64/tx_float_neon.S
index 255dc3acaa..2db0e95740 100644
--- a/libavutil/aarch64/tx_float_neon.S
+++ b/libavutil/aarch64/tx_float_neon.S
@@ -671,6 +671,105 @@ endfunc
 FFT15_FN float,    0
 FFT15_FN ns_float, 1

+// 15xM PFA (len = 15*M, M a power of two), like the x86 fft_pfa_15xM:
+// dim1 = M 15pt transforms scattered into s->tmp at sub_map[i], spaced M
+// apart; dim2 = 15 in-place M-pt subtransforms (plain blr, the uniform ABI
+// needs no asm-call variant); post = out[i*stride] = s->tmp[out_map[i]].
+// AVTXContext offsets: len=0, map=8, tmp=24, sub=32, fn[0]=40.
+.macro PFA_15_FN name, no_perm
+function ff_tx_fft_pfa_15xM_\name\()_neon, export=1
+        AARCH64_SIGN_LINK_REGISTER
+        stp             x29, x30, [sp, #-128]!
+        mov             x29, sp
+        stp             x19, x20, [sp, #16]
+        stp             x21, x22, [sp, #32]
+        stp             x23, x24, [sp, #48]
+        stp             x25, x26, [sp, #64]
+        stp             x27, x28, [sp, #80]
+        stp             d8,  d9,  [sp, #96]
+        str             d10, [sp, #112]
+
+        mov             x25, x0                     // root context
+        mov             x26, x1                     // user out
+        mov             x27, x3                     // user stride (bytes)
+
+        ldr             w19, [x0, #0]               // len = 15*M
+        ldr             x20, [x0, #24]              // s->tmp
+        ldr             x23, [x0, #32]              // s->sub
+        ldr             w24, [x23, #0]              // M
+        ldr             x28, [x23, #8]              // sub_map
+        lsl             x3,  x24, #3                // M*8 = 15pt output stride
+.if \no_perm == 0
+        ldr             x4,  [x0, #8]               // in_map, advanced by the loads
+.endif
+
+        movrel          x5, X(ff_tx_tab_53_float)
+        ld1             { v28.4s, v29.4s, v30.4s }, [x5]
+        movrel          x5, tab_15pt
+        ld1             { v24.4s }, [x5]
+        LOAD_SUBADD
+        FFT15_DERIVE_CONSTS
+        add             x6,  x3, x3, lsl #1         // stride*3
+        add             x7,  x3, x3, lsl #2         // stride*5
+1:
+        ldr             w10, [x28], #4              // sub_map[i]
+        add             x1,  x20, x10, lsl #3
+        FFT15_LOAD      \no_perm, advance=1
+        FFT15_CORE      hoist_strides=1
+        subs            w19, w19, #15
+        b.gt            1b
+
+        ldr             x28, [x25, #40]             // ctx->fn[0]
+        mov             x21, x20                    // column base
+        mov             w19, #15
+2:
+        mov             x0,  x23
+        mov             x1,  x21
+        mov             x2,  x21
+        mov             x3,  #8
+        blr             x28                         // M-point FFT, in-place
+        add             x21, x21, x24, lsl #3
+        subs            w19, w19, #1
+        b.gt            2b
+
+        // out[i*stride] = s->tmp[out_map[i]], unrolled by 4 (len % 60 == 0)
+        ldr             w19, [x25, #0]              // len
+        ldr             x22, [x25, #8]              // s->map
+        add             x22, x22, x19, lsl #2       // out_map = map + len
+        mov             x10, x26
+        lsl             x11, x27, #1                // 2*stride
+        add             x12, x27, x11               // 3*stride
+3:
+        ldp             w13, w14, [x22], #8         // out_map[i,   i+1]
+        ldp             w15, w16, [x22], #8         // out_map[i+2, i+3]
+        ldr             d0,  [x20, x13, lsl #3]
+        ldr             d1,  [x20, x14, lsl #3]
+        ldr             d2,  [x20, x15, lsl #3]
+        ldr             d3,  [x20, x16, lsl #3]
+        str             d0,  [x10]
+        str             d1,  [x10, x27]
+        str             d2,  [x10, x11]
+        str             d3,  [x10, x12]
+        add             x10, x10, x11, lsl #1
+        subs            w19, w19, #4
+        b.gt            3b
+
+        ldp             x19, x20, [sp, #16]
+        ldp             x21, x22, [sp, #32]
+        ldp             x23, x24, [sp, #48]
+        ldp             x25, x26, [sp, #64]
+        ldp             x27, x28, [sp, #80]
+        ldp             d8,  d9,  [sp, #96]
+        ldr             d10, [sp, #112]
+        ldp             x29, x30, [sp], #128
+        AARCH64_VALIDATE_LINK_REGISTER
+        ret
+endfunc
+.endm
+
+PFA_15_FN float,    0
+PFA_15_FN ns_float, 1
+
 .macro SETUP_SR_RECOMB len, re, im, dec
         ldr             w5, =(\len - 4*7)
         movrel          \re, X(ff_tx_tab_\len\()_float)