Commit d1fd9c7b9c for ffmpeg

commit d1fd9c7b9cc5fc3c165115264e2852ac6e075906
Author: Ramiro Polla <ramiro.polla@gmail.com>
Date:   Sun Jul 19 13:22:54 2026 +0200

    swscale/aarch64/ops_asmgen: implement integer linear transformations

    checkasm:
     - CPU: ARM Cortex-A720
     - Timing source: linux (perf)
    Benchmark results:
      name                                                     ticks (vs ref)
      u8_linear_x_x000x_c:                                    1117.8
      u8_linear_x_x000x_aarch64_neon:                          463.6 ( 2.41x)
      u8_linear_xyz_x000x_0x00x_00x0x_c:                      1269.2
      u8_linear_xyz_x000x_0x00x_00x0x_aarch64_neon:            586.3 ( 2.16x)
      u8_linear_xyz_x0000_0x000_00x00_c:                      1270.5
      u8_linear_xyz_x0000_0x000_00x00_aarch64_neon:            503.3 ( 2.52x)
      u8_linear_y_0x000_c:                                    1274.3
      u8_linear_y_0x000_aarch64_neon:                          443.8 ( 2.87x)
      u16_linear_x_x000x_c:                                   1627.6
      u16_linear_x_x000x_aarch64_neon:                         565.8 ( 2.88x)
      u16_linear_xyz_x000x_0x00x_00x0x_c:                     1807.0
      u16_linear_xyz_x000x_0x00x_00x0x_aarch64_neon:           830.5 ( 2.18x)
      u16_linear_xyz_x0000_0x000_00x00_c:                     1686.4
      u16_linear_xyz_x0000_0x000_00x00_aarch64_neon:           585.9 ( 2.88x)
      u16_linear_xyzw_x0000_0x000_00x00_000x0_c:              1753.8
      u16_linear_xyzw_x0000_0x000_00x00_000x0_aarch64_neon:    680.4 ( 2.58x)
      u32_linear_x_x000x_c:                                   2617.0
      u32_linear_x_x000x_aarch64_neon:                        1007.5 ( 2.60x)
      u32_linear_xyz_x000x_0x00x_00x0x_c:                     3107.7
      u32_linear_xyz_x000x_0x00x_00x0x_aarch64_neon:          1504.7 ( 2.06x)
      u32_linear_xyz_x0000_0x000_00x00_c:                     3106.0
      u32_linear_xyz_x0000_0x000_00x00_aarch64_neon:          1008.3 ( 3.08x)

    On in-order cores (Cortex-A520), checkasm reports NEON slower than C for
    most op chains. This is mainly an artifact of the checkasm buffer layout:
    the input and output planes are 16 KiB apart and conflict in the 4-way
    32 KiB L1D, and the smaller NEON block sizes (8/16 vs 32) touch each
    cache line more often. In swscale benchmarks on the same core, the
    conversions that now use integer linear ops are 1.1x-2.9x faster than
    with the previous f32 NEON path.

    Sponsored-by: Sovereign Tech Fund
    Signed-off-by: Ramiro Polla <ramiro.polla@gmail.com>

diff --git a/libswscale/aarch64/ops.c b/libswscale/aarch64/ops.c
index b02312b4b7..f1d3d698ab 100644
--- a/libswscale/aarch64/ops.c
+++ b/libswscale/aarch64/ops.c
@@ -74,7 +74,7 @@ static int aarch64_setup_linear(const SwsAArch64OpImplParams *p,
      */
     const int num_vregs = linear_num_vregs(p);
     av_assert0(num_vregs <= 4);
-    float *coeffs = av_malloc(num_vregs * 4 * sizeof(float));
+    void *coeffs = av_malloc(num_vregs * 16);
     if (!coeffs)
         return AVERROR(ENOMEM);

@@ -86,8 +86,15 @@ static int aarch64_setup_linear(const SwsAArch64OpImplParams *p,
     for (int i = 0; i < 4; i++) {
         for (int j = 0; j < 5; j++) {
             const int jj = (j == 0) ? 4 : (j - 1);
-            if (!(p->par.lin.zero & SWS_MASK(i, jj)))
-                coeffs[i_coeff++] = uop->data.mat4[i][jj].f32;
+            if (!(p->par.lin.zero & SWS_MASK(i, jj))) {
+                const SwsPixel px = uop->data.mat4[i][jj];
+                switch (p->type) {
+                case SWS_PIXEL_U8:  ((uint8_t  *) coeffs)[i_coeff++] = px.u8;  break;
+                case SWS_PIXEL_U16: ((uint16_t *) coeffs)[i_coeff++] = px.u16; break;
+                case SWS_PIXEL_U32: ((uint32_t *) coeffs)[i_coeff++] = px.u32; break;
+                case SWS_PIXEL_F32: ((float    *) coeffs)[i_coeff++] = px.f32; break;
+                }
+            }
         }
     }

diff --git a/libswscale/aarch64/ops_asmgen.c b/libswscale/aarch64/ops_asmgen.c
index f83e487bc7..b2eb0b178d 100644
--- a/libswscale/aarch64/ops_asmgen.c
+++ b/libswscale/aarch64/ops_asmgen.c
@@ -875,9 +875,14 @@ static void linear_pass(SwsAArch64Context *s, const SwsAArch64OpImplParams *p,
             } else if (first && !is_offset) {
                 if (p->par.lin.one & SWS_MASK(i, src_j)) {
                     i_mov16b(r, dx[i], vsrc);           CMTF("v%c[%u]  = vsrc%c[%u];", cvh, i, cvh, src_j);
-                } else {
+                } else if (p->type == SWS_PIXEL_F32) {
                     i_fmul  (r, dx[i], vsrc, vcoeff);   CMTF("v%c[%u]  = vsrc%c[%u] * coeff[%u][%u];", cvh, i, cvh, src_j, i, src_j);
+                } else {
+                    i_mul   (r, dx[i], vsrc, vcoeff);   CMTF("v%c[%u]  = vsrc%c[%u] * coeff[%u][%u];", cvh, i, cvh, src_j, i, src_j);
                 }
+            } else if (p->type != SWS_PIXEL_F32) {
+                /* Integer multiply-accumulate is always exact. */
+                i_mla (r, dx[i], vsrc, vcoeff);         CMTF("v%c[%u] += vsrc%c[%u] * coeff[%u][%u];", cvh, i, cvh, src_j, i, src_j);
             } else if (p->uop == SWS_UOP_LINEAR_FMA) {
                 /**
                  * Most modern aarch64 cores have a fastpath for sequences
diff --git a/libswscale/aarch64/ops_impl.h b/libswscale/aarch64/ops_impl.h
index 560021e4ce..c9ad019cfe 100644
--- a/libswscale/aarch64/ops_impl.h
+++ b/libswscale/aarch64/ops_impl.h
@@ -63,6 +63,12 @@ typedef struct SwsAArch64OpImplParams {
 #define LOOP_MASK(p, idx) LOOP(p->mask, idx)
 #define LOOP_MASK_BWD(p, idx) LOOP_BWD(p->mask, idx)

+/* Number of elements in each linear coefficient vector register. */
+static inline int linear_vreg_nelems(SwsPixelType type)
+{
+    return 16 / ff_sws_pixel_type_size(type);
+}
+
 /* Compute number of vector registers needed to store all coefficients. */
 static inline int linear_num_vregs(const SwsAArch64OpImplParams *params)
 {
@@ -70,7 +76,8 @@ static inline int linear_num_vregs(const SwsAArch64OpImplParams *params)
     for (int i = 0; i < 4 * 5; i++)
         if (!(params->par.lin.zero & (1ULL << i)))
             count++;
-    return (count + 3) / 4;
+    const int nelems = linear_vreg_nelems(params->type);
+    return (count + nelems - 1) / nelems;
 }

 /**
diff --git a/libswscale/aarch64/ops_static.c b/libswscale/aarch64/ops_static.c
index 74891d2be3..2a7933d675 100644
--- a/libswscale/aarch64/ops_static.c
+++ b/libswscale/aarch64/ops_static.c
@@ -307,6 +307,19 @@ static void asmgen_setup_linear(SwsAArch64Context *s, const SwsAArch64OpImplPara
     asmgen_set_load_cont_node(s);
     i_ld1(r, coeff_veclist, a64op_base(ptr));               CMT("coeff_veclist = *vcoeff_ptr;");

+    /**
+     * Integer mul/mla by element does not exist for 8-bit elements, and
+     * requires the element register to be in v0-v15 for 16-bit elements.
+     * For these types, the coefficients used as multiplication operands
+     * are broadcast into full temp vectors instead. Temp vectors 8-11
+     * are otherwise only used by the floating-point fmul+fadd path, so
+     * they are free for integer types.
+     */
+    const bool dup_coeffs = (p->type == SWS_PIXEL_U8 || p->type == SWS_PIXEL_U16);
+    const int nelems = linear_vreg_nelems(p->type);
+    RasmOp *vint = &vt[8];
+    int i_vint = 0;
+
     /**
      * Populate operands matrix from packed data into linear_vcoeff matrix
      * and compute mask for rows that must be saved before being overwritten.
@@ -315,15 +328,28 @@ static void asmgen_setup_linear(SwsAArch64Context *s, const SwsAArch64OpImplPara
     bool overwritten[4] = { false, false, false, false };
     int i_coeff = 0;
     LOOP_MASK(p, i) {
+        bool first = true;
         for (int j = 0; j < 5; j++) {
             bool is_offset = (j == 0);
             int src_j = is_offset ? 4 : (j - 1);
             if (p->par.lin.zero & SWS_MASK(i, src_j))
                 continue;
-            uint8_t vc_i = i_coeff / 4;
-            uint8_t vc_j = i_coeff & 3;
-            regs->linear_vcoeff[i][j] = a64op_elem(vc[vc_i], vc_j);
+            uint8_t vc_i = i_coeff / nelems;
+            uint8_t vc_j = i_coeff & (nelems - 1);
+            RasmOp vcoeff = a64op_elem(vc[vc_i], vc_j);
+            /**
+             * The first coefficient of a row is not used as a multiplication
+             * operand if it is either an offset (broadcast) or one (move).
+             */
+            bool is_one = !is_offset && (p->par.lin.one & SWS_MASK(i, src_j));
+            if (dup_coeffs && !(first && (is_offset || is_one))) {
+                av_assert0(i_vint < 4);
+                i_dup(r, vint[i_vint], vcoeff);             CMTF("vcoeff%u = broadcast(coeff[%u][%u]);", i_vint, i, src_j);
+                vcoeff = vint[i_vint++];
+            }
+            regs->linear_vcoeff[i][j] = vcoeff;
             i_coeff++;
+            first = false;
             if (!is_offset && overwritten[src_j])
                 save_mask |= SWS_COMP(src_j);
             overwritten[i] = true;
diff --git a/libswscale/tests/sws_ops_aarch64.c b/libswscale/tests/sws_ops_aarch64.c
index 6ce1ab67f4..05eba0f0bb 100644
--- a/libswscale/tests/sws_ops_aarch64.c
+++ b/libswscale/tests/sws_ops_aarch64.c
@@ -219,7 +219,7 @@ static int collect_ops_compile(SwsContext *ctx, const SwsOpList *ops,
         ret = aarch64_collect_op(&params, root);
         if (ret < 0)
             goto end;
-        if (params.uop == SWS_UOP_LINEAR_FMA) {
+        if (params.uop == SWS_UOP_LINEAR_FMA && params.type == SWS_PIXEL_F32) {
             /**
              * Generate both sets of linear op functions that do use
              * and do not use fmla (selected by SWS_BITEXACT).