Commit 05ecb86e31 for ffmpeg

commit 05ecb86e31e89ad82ca084793971680cf2ef58b2
Author: Ramiro Polla <ramiro.polla@gmail.com>
Date:   Tue Sep 29 02:51:47 2026 +0200

    swscale/aarch64/ops: port to uops

    Translate the SwsOpList to uops with ff_sws_ops_translate() and build
    the aarch64 implementation parameters directly from the resulting
    SwsUOps, instead of mapping each SwsOp to an uop in the backend.

    This removes the SwsOp-specific logic from ops_impl_conv.c (read/write
    mode selection, swizzle to move decomposition, clear/linear/dither
    parameter extraction), and uses the generic ff_sws_setup_vec4() and
    ff_sws_setup_scalar() helpers for clear, min, max and scale. The dither
    matrix, including the extra rows for over-reading with y offsets, is
    now prepared by the uops layer, so the backend only takes a reference
    to it.

    The backend also implements compile_uops, so it is now covered by the
    sw_ops checkasm test.

    One pair of copy entries in ops_entries.c changes because the uops
    layer decomposes that swizzle into a different (and better) sequence
    of moves.

    Sponsored-by: Sovereign Tech Fund
    Signed-off-by: Ramiro Polla <ramiro.polla@gmail.com>

diff --git a/libswscale/aarch64/ops.c b/libswscale/aarch64/ops.c
index 96058a282f..b02312b4b7 100644
--- a/libswscale/aarch64/ops.c
+++ b/libswscale/aarch64/ops.c
@@ -66,7 +66,7 @@ static SwsFuncPtr aarch64_lookup(const SwsAArch64OpImplParams *p)

 /*********************************************************************/
 static int aarch64_setup_linear(const SwsAArch64OpImplParams *p,
-                                const SwsOp *op, SwsImplResult *res)
+                                const SwsUOp *uop, SwsImplResult *res)
 {
     /**
      * Compute number of full vector registers needed to pack all non-zero
@@ -87,7 +87,7 @@ static int aarch64_setup_linear(const SwsAArch64OpImplParams *p,
         for (int j = 0; j < 5; j++) {
             const int jj = (j == 0) ? 4 : (j - 1);
             if (!(p->par.lin.zero & SWS_MASK(i, jj)))
-                coeffs[i_coeff++] = (float) op->lin.m[i][jj].num / op->lin.m[i][jj].den;
+                coeffs[i_coeff++] = uop->data.mat4[i][jj].f32;
         }
     }

@@ -98,103 +98,54 @@ static int aarch64_setup_linear(const SwsAArch64OpImplParams *p,
 }

 /*********************************************************************/
-static int aarch64_setup_dither(const SwsAArch64OpImplParams *p,
-                                const SwsOp *op, SwsImplResult *res)
+static int aarch64_setup_dither(const SwsUOp *uop, SwsImplResult *res)
 {
-    /**
-     * The input dither matrix is (1 << size_log2)² pixels large. It is
-     * periodic, so the x and y offsets should be masked to fit inside
-     * (1 << size_log2).
-     * The width of the matrix is assumed to be at least 8, which matches
-     * the maximum block_size for aarch64 asmgen when f32 operations
-     * (i.e., dithering) are used. This guarantees that the x offset is
-     * aligned and that reading block_size elements does not extend past
-     * the end of the row. The x offset doesn't change between components,
-     * so it is only required to be masked once.
-     * The y offset, on the other hand, may change per component, and
-     * would therefore need to be masked for every y_offset value. To
-     * simplify the execution, we over-allocate the number of rows of
-     * the output dither matrix by the largest y_offset value. This way,
-     * we only need to mask y offset once, and can safely increment the
-     * dither matrix pointer by fixed offsets for every y_offset change.
-     */
-
-    /* Find the largest y_offset value. */
-    const int size = 1 << op->dither.size_log2;
-    const int8_t *off = op->dither.y_offset;
-    int max_offset = 0;
-    for (int i = 0; i < 4; i++) {
-        if (off[i] >= 0)
-            max_offset = FFMAX(max_offset, off[i] & (size - 1));
-    }
-
-    /* Allocate (size + max_offset) rows to allow over-reading the matrix. */
-    const int stride = size * sizeof(float);
-    const int num_rows = size + max_offset;
-    float *matrix = av_malloc(num_rows * stride);
-    if (!matrix)
-        return AVERROR(ENOMEM);
-
-    for (int i = 0; i < size * size; i++)
-        matrix[i] = (float) op->dither.matrix[i].num / op->dither.matrix[i].den;
-
-    memcpy(&matrix[size * size], matrix, max_offset * stride);
-
-    res->priv.ptr = matrix;
-    res->free = ff_op_priv_free;
-
+    res->priv.ptr = av_refstruct_ref(uop->data.ptr);
+    res->free = ff_op_priv_unref;
     return 0;
 }

 /*********************************************************************/
-static int aarch64_setup(const SwsOpList *ops, int block_size, int n,
-                         const SwsAArch64OpImplParams *p, SwsImplResult *out)
+static int aarch64_setup(const SwsUOp *uop, const SwsAArch64OpImplParams *p,
+                         SwsImplResult *out)
 {
-    const SwsOp *op = &ops->ops[n];
-    switch (op->op) {
-    case SWS_OP_READ:
+    switch (uop->uop) {
+    case SWS_UOP_READ_BIT:
         /* Negative shift values to perform right shift using ushl. */
-        if (op->rw.frac == 3) {
-            out->priv = (SwsOpPriv) {
-                .u8 = {
-                    -7, -6, -5, -4, -3, -2, -1, 0,
-                    -7, -6, -5, -4, -3, -2, -1, 0,
-                }
-            };
-        }
+        out->priv = (SwsOpPriv) {
+            .u8 = {
+                -7, -6, -5, -4, -3, -2, -1, 0,
+                -7, -6, -5, -4, -3, -2, -1, 0,
+            }
+        };
         break;
-    case SWS_OP_WRITE:
+    case SWS_UOP_WRITE_BIT:
         /* Shift values for ushl. */
-        if (op->rw.frac == 3) {
-            out->priv = (SwsOpPriv) {
-                .u8 = {
-                    7, 6, 5, 4, 3, 2, 1, 0,
-                    7, 6, 5, 4, 3, 2, 1, 0,
-                }
-            };
-        }
+        out->priv = (SwsOpPriv) {
+            .u8 = {
+                7, 6, 5, 4, 3, 2, 1, 0,
+                7, 6, 5, 4, 3, 2, 1, 0,
+            }
+        };
         break;
-    case SWS_OP_CLEAR:
-        ff_sws_setup_clear(&(const SwsImplParams) { .op = op }, out);
-        break;
-    case SWS_OP_MIN:
-    case SWS_OP_MAX:
-        ff_sws_setup_clamp(&(const SwsImplParams) { .op = op }, out);
-        break;
-    case SWS_OP_SCALE:
-        ff_sws_setup_scale(&(const SwsImplParams) { .op = op }, out);
-        break;
-    case SWS_OP_LINEAR:
-        return aarch64_setup_linear(p, op, out);
-    case SWS_OP_DITHER:
-        return aarch64_setup_dither(p, op, out);
+    case SWS_UOP_CLEAR:
+    case SWS_UOP_MIN:
+    case SWS_UOP_MAX:
+        return ff_sws_setup_vec4(&(const SwsImplParams) { .uop = uop }, out);
+    case SWS_UOP_SCALE:
+        return ff_sws_setup_scalar(&(const SwsImplParams) { .uop = uop }, out);
+    case SWS_UOP_LINEAR:
+    case SWS_UOP_LINEAR_FMA:
+        return aarch64_setup_linear(p, uop, out);
+    case SWS_UOP_DITHER:
+        return aarch64_setup_dither(uop, out);
     }
     return 0;
 }

 /*********************************************************************/
-static int aarch64_compile(SwsContext *ctx, const SwsOpList *ops,
-                           SwsCompiledOp *out)
+static int aarch64_compile_uops(SwsContext *ctx, const SwsUOpList *uops,
+                                SwsCompiledOp *out)
 {
     int ret;

@@ -203,7 +154,7 @@ static int aarch64_compile(SwsContext *ctx, const SwsOpList *ops,
         return AVERROR(ENOTSUP);

     /* Use at most two full vregs during the widest precision section */
-    int block_size = (ff_sws_op_list_max_size(ops) == 4) ? 8 : 16;
+    int block_size = (uops->pixel_size_max == 4) ? 8 : 16;

     SwsOpChain *chain = ff_sws_op_chain_alloc();
     if (!chain)
@@ -218,18 +169,16 @@ static int aarch64_compile(SwsContext *ctx, const SwsOpList *ops,
     };

     /* Look up kernel functions. */
-    for (int i = 0; i < ops->num_ops; i++) {
+    for (int i = 0; i < uops->num_ops; i++) {
         SwsAArch64OpImplParams params = { 0 };
-        ret = convert_to_aarch64_impl(ctx, ops, i, block_size, &params);
-        if (ret < 0)
-            goto error;
+        convert_to_aarch64_impl(&uops->ops[i], block_size, &params);
         SwsFuncPtr func = aarch64_lookup(&params);
         if (!func) {
             ret = AVERROR(ENOTSUP);
             goto error;
         }
         SwsImplResult res = { 0 };
-        ret = aarch64_setup(ops, block_size, i, &params, &res);
+        ret = aarch64_setup(&uops->ops[i], &params, &res);
         if (ret < 0)
             goto error;
         ret = ff_sws_op_chain_append(chain, func, res.free, &res.priv);
@@ -243,12 +192,8 @@ static int aarch64_compile(SwsContext *ctx, const SwsOpList *ops,
     void ff_sws_process_0111_neon(void);
     void ff_sws_process_1111_neon(void);

-    const SwsOp *read  = ff_sws_op_list_input(ops);
-    const SwsOp *write = ff_sws_op_list_output(ops);
-    const int read_planes  = read ? ff_sws_rw_op_planes(read) : 0;
-    const int write_planes = ff_sws_rw_op_planes(write);
     SwsOpFunc process_func = NULL;
-    switch (FFMAX(read_planes, write_planes)) {
+    switch (av_popcount(uops->planes_in | uops->planes_out)) {
     case 1: process_func = (SwsOpFunc) ff_sws_process_0001_neon; break;
     case 2: process_func = (SwsOpFunc) ff_sws_process_0011_neon; break;
     case 3: process_func = (SwsOpFunc) ff_sws_process_0111_neon; break;
@@ -258,16 +203,38 @@ static int aarch64_compile(SwsContext *ctx, const SwsOpList *ops,
     out->func      = process_func;
     out->cpu_flags = chain->cpu_flags;

+    return 0;
+
 error:
+    ff_sws_op_chain_free(chain);
+    return ret;
+}
+
+/*********************************************************************/
+static int aarch64_compile(SwsContext *ctx, const SwsOpList *ops,
+                           SwsCompiledOp *out)
+{
+    SwsUOpList *uops = ff_sws_uop_list_alloc();
+    if (!uops)
+        return AVERROR(ENOMEM);
+
+    const SwsUOpFlags flags = (ctx->flags & SWS_BITEXACT) ? 0 : SWS_UOP_FLAG_FMA;
+    int ret = ff_sws_ops_translate(ctx, ops, flags, uops);
     if (ret < 0)
-        ff_sws_op_chain_free(chain);
+        goto error;
+
+    ret = aarch64_compile_uops(ctx, uops, out);
+
+error:
+    ff_sws_uop_list_free(&uops);
     return ret;
 }

 /*********************************************************************/
 const SwsOpBackend backend_aarch64 = {
-    .name      = "aarch64",
-    .flags     = SWS_BACKEND_AARCH64,
-    .compile   = aarch64_compile,
-    .hw_format = AV_PIX_FMT_NONE,
+    .name         = "aarch64",
+    .flags        = SWS_BACKEND_AARCH64,
+    .compile      = aarch64_compile,
+    .compile_uops = aarch64_compile_uops,
+    .hw_format    = AV_PIX_FMT_NONE,
 };
diff --git a/libswscale/aarch64/ops_asmgen.c b/libswscale/aarch64/ops_asmgen.c
index 6068fc327a..f83e487bc7 100644
--- a/libswscale/aarch64/ops_asmgen.c
+++ b/libswscale/aarch64/ops_asmgen.c
@@ -938,8 +938,23 @@ static void asmgen_op_dither(SwsAArch64Context *s, const SwsAArch64OpImplParams
     RasmOp y64 = a64op_x(s->y);

     /**
-     * For a description of the matrix buffer layout, read the comments
-     * in aarch64_setup_dither() in aarch64/ops.c.
+     * The dither matrix is (1 << size_log2)² pixels large. It is
+     * periodic, so the x and y offsets should be masked to fit inside
+     * (1 << size_log2). The matrix buffer is prepared by
+     * translate_dither_op() in libswscale/uops.c.
+     * The width of the matrix is assumed to be at least 8, which matches
+     * the maximum block_size for aarch64 asmgen when f32 operations
+     * (i.e., dithering) are used. This guarantees that the x offset is
+     * aligned and that reading block_size elements does not extend past
+     * the end of the row. The x offset doesn't change between components,
+     * so it is only required to be masked once.
+     * The y offset, on the other hand, may change per component, and
+     * would therefore need to be masked for every y_offset value. To
+     * avoid this, the matrix buffer is over-allocated by the largest
+     * y_offset value, with the extra rows repeating the first rows of
+     * the matrix. This way, we only need to mask the y offset once, and
+     * can safely increment the dither matrix pointer by fixed offsets
+     * for every y_offset change.
      */

     /**
diff --git a/libswscale/aarch64/ops_entries.c b/libswscale/aarch64/ops_entries.c
index 6bb9db1613..cd42184715 100644
--- a/libswscale/aarch64/ops_entries.c
+++ b/libswscale/aarch64/ops_entries.c
@@ -159,8 +159,8 @@ ENTRY(ff_sws_copy_000000103120_32_u8_1110_neon,     { .uop = SWS_UOP_COPY,
 ENTRY(ff_sws_copy_000000302010_8_u8_1110_neon,      { .uop = SWS_UOP_COPY,         .block_size =  8, .type = SWS_PIXEL_U8,  .mask = 0xe, .par.move = { .num_moves = 3, .dst = {1, 2, 3, 0, 0, 0}, .src = {0, 0, 0, 0, 0, 0} } })
 ENTRY(ff_sws_copy_000000302010_16_u8_1110_neon,     { .uop = SWS_UOP_COPY,         .block_size = 16, .type = SWS_PIXEL_U8,  .mask = 0xe, .par.move = { .num_moves = 3, .dst = {1, 2, 3, 0, 0, 0}, .src = {0, 0, 0, 0, 0, 0} } })
 ENTRY(ff_sws_copy_000000302010_32_u8_1110_neon,     { .uop = SWS_UOP_COPY,         .block_size = 32, .type = SWS_PIXEL_U8,  .mask = 0xe, .par.move = { .num_moves = 3, .dst = {1, 2, 3, 0, 0, 0}, .src = {0, 0, 0, 0, 0, 0} } })
-ENTRY(ff_sws_copy_001f01f03020_8_u8_1111_neon,      { .uop = SWS_UOP_COPY,         .block_size =  8, .type = SWS_PIXEL_U8,  .mask = 0xf, .par.move = { .num_moves = 5, .dst = {2, 3, -1, 0, 1, 0}, .src = {0, 0, 0, 1, -1, 0} } })
-ENTRY(ff_sws_copy_001f01f03020_16_u8_1111_neon,     { .uop = SWS_UOP_COPY,         .block_size = 16, .type = SWS_PIXEL_U8,  .mask = 0xf, .par.move = { .num_moves = 5, .dst = {2, 3, -1, 0, 1, 0}, .src = {0, 0, 0, 1, -1, 0} } })
+ENTRY(ff_sws_copy_000032120120_8_u8_1111_neon,      { .uop = SWS_UOP_COPY,         .block_size =  8, .type = SWS_PIXEL_U8,  .mask = 0xf, .par.move = { .num_moves = 4, .dst = {2, 0, 1, 3, 0, 0}, .src = {0, 1, 2, 2, 0, 0} } })
+ENTRY(ff_sws_copy_000032120120_16_u8_1111_neon,     { .uop = SWS_UOP_COPY,         .block_size = 16, .type = SWS_PIXEL_U8,  .mask = 0xf, .par.move = { .num_moves = 4, .dst = {2, 0, 1, 3, 0, 0}, .src = {0, 1, 2, 2, 0, 0} } })
 ENTRY(ff_sws_swap_bytes_8_u16_0001_neon,            { .uop = SWS_UOP_SWAP_BYTES,   .block_size =  8, .type = SWS_PIXEL_U16, .mask = 0x1 })
 ENTRY(ff_sws_swap_bytes_8_u16_0010_neon,            { .uop = SWS_UOP_SWAP_BYTES,   .block_size =  8, .type = SWS_PIXEL_U16, .mask = 0x2 })
 ENTRY(ff_sws_swap_bytes_8_u16_0011_neon,            { .uop = SWS_UOP_SWAP_BYTES,   .block_size =  8, .type = SWS_PIXEL_U16, .mask = 0x3 })
diff --git a/libswscale/aarch64/ops_impl.h b/libswscale/aarch64/ops_impl.h
index 304b68e44c..560021e4ce 100644
--- a/libswscale/aarch64/ops_impl.h
+++ b/libswscale/aarch64/ops_impl.h
@@ -41,7 +41,7 @@ static inline uint16_t nibble_mask(SwsCompMask mask)

 /**
  * SwsAArch64OpImplParams describes the parameters for an SwsUOpType
- * operation. It consists of simplified parameters from the SwsOp structure,
+ * operation. It consists of simplified parameters from the SwsUOp structure,
  * with the purpose of being straight-forward to implement and execute.
  */
 typedef struct SwsAArch64OpImplParams {
diff --git a/libswscale/aarch64/ops_impl_conv.c b/libswscale/aarch64/ops_impl_conv.c
index 9869c29608..4f8e6e3437 100644
--- a/libswscale/aarch64/ops_impl_conv.c
+++ b/libswscale/aarch64/ops_impl_conv.c
@@ -24,297 +24,47 @@
  */

 #include "libavutil/error.h"
-#include "libavutil/rational.h"
-#include "libswscale/ops.h"

 #include "ops_impl.h"

-static void swizzle_emit(SwsAArch64OpImplParams *out, uint8_t dst, uint8_t src)
-{
-    int idx = out->par.move.num_moves++;
-    out->par.move.dst[idx] = dst;
-    out->par.move.src[idx] = src;
-}
-
-static void convert_swizzle_to_moves(const SwsOp *op, SwsAArch64OpImplParams *out)
-{
-    SwsSwizzleOp swizzle = {
-        .in = {
-            op->swizzle.in[0],
-            op->swizzle.in[1],
-            op->swizzle.in[2],
-            op->swizzle.in[3],
-        }
-    };
-
-    /* Compute used vectors (src and dst) */
-    uint8_t src_used[4] = { 0 };
-    bool done[4] = { true, true, true, true };
-    LOOP(out->mask, dst) {
-        uint8_t src = swizzle.in[dst];
-        src_used[src]++;
-        done[dst] = false;
-    }
-
-    /* First perform unobstructed copies. */
-    for (bool progress = true; progress; ) {
-        progress = false;
-        for (int dst = 0; dst < 4; dst++) {
-            if (done[dst] || src_used[dst])
-                continue;
-            uint8_t src = swizzle.in[dst];
-            swizzle_emit(out, dst, src);
-            src_used[src]--;
-            done[dst] = true;
-            progress = true;
-        }
-    }
-
-    /* Then swap and rotate remaining operations. */
-    for (int dst = 0; dst < 4; dst++) {
-        if (done[dst])
-            continue;
-
-        swizzle_emit(out, -1, dst);
-
-        uint8_t cur_dst = dst;
-        uint8_t src = swizzle.in[cur_dst];
-        while (src != dst) {
-            swizzle_emit(out, cur_dst, src);
-            done[cur_dst] = true;
-            cur_dst = src;
-            src = swizzle.in[cur_dst];
-        }
-
-        swizzle_emit(out, cur_dst, -1);
-        done[cur_dst] = true;
-    }
-}
-
 /**
- * Convert SwsOp to a SwsAArch64OpImplParams. Read the comments regarding
+ * Convert SwsUOp to a SwsAArch64OpImplParams. Read the comments regarding
  * SwsAArch64OpImplParams in ops_impl.h for more information.
  */
-static int convert_to_aarch64_impl(SwsContext *ctx, const SwsOpList *ops, int n,
-                                   int block_size, SwsAArch64OpImplParams *out)
+static void convert_to_aarch64_impl(const SwsUOp *uop, int block_size,
+                                    SwsAArch64OpImplParams *out)
 {
-    const SwsOp *op = &ops->ops[n];
-
+    out->uop = uop->uop;
+    out->mask = uop->mask;
+    out->type = uop->type;
     out->block_size = block_size;
+    out->par = uop->par;

     /**
-     * Most SwsOp work on fields described by SWS_OP_NEEDED().
-     * The few that don't will override this field later.
+     * Deduplicate params to prevent identical CPS functions from being
+     * instantiated multiple times under different names.
      */
-    out->mask = 0;
-    for (int i = 0; i < 4; i++) {
-        if (SWS_OP_NEEDED(op, i))
-            out->mask |= SWS_COMP(i);
-    }
-
-    out->type = op->type;
-
-    /* Map SwsOpType to SwsUOpType */
-    switch (op->op) {
-    case SWS_OP_READ:
-        if (op->rw.filter.op)
-            return AVERROR(ENOTSUP);
-        /**
-         * The different types of read operations have been split into
-         * their own SwsUOpType to simplify the implementation.
-         */
-        if (op->rw.frac == 1)
-            out->uop = SWS_UOP_READ_NIBBLE;
-        else if (op->rw.frac == 3)
-            out->uop = SWS_UOP_READ_BIT;
-        else if (op->rw.mode == SWS_RW_PACKED && op->rw.elems > 1)
-            out->uop = SWS_UOP_READ_PACKED;
-        else if (op->rw.mode == SWS_RW_PACKED || op->rw.mode == SWS_RW_PLANAR)
-            out->uop = SWS_UOP_READ_PLANAR;
-        else
-            return AVERROR(ENOTSUP);
-        break;
-    case SWS_OP_WRITE:
-        if (op->rw.filter.op)
-            return AVERROR(ENOTSUP);
-        /**
-         * The different types of write operations have been split into
-         * their own SwsUOpType to simplify the implementation.
-         */
-        if (op->rw.frac == 1)
-            out->uop = SWS_UOP_WRITE_NIBBLE;
-        else if (op->rw.frac == 3)
-            out->uop = SWS_UOP_WRITE_BIT;
-        else if (op->rw.mode == SWS_RW_PACKED && op->rw.elems > 1)
-            out->uop = SWS_UOP_WRITE_PACKED;
-        else if (op->rw.mode == SWS_RW_PACKED || op->rw.mode == SWS_RW_PLANAR)
-            out->uop = SWS_UOP_WRITE_PLANAR;
-        else
-            return AVERROR(ENOTSUP);
-        break;
-    case SWS_OP_SWAP_BYTES: out->uop = SWS_UOP_SWAP_BYTES; break;
-    case SWS_OP_SWIZZLE: {
-        /**
-         * Detect whether copies are needed or if a simple permute is
-         * enough.
-         */
-        out->uop = SWS_UOP_PERMUTE;
-        SwsCompMask seen = 0;
-        LOOP(out->mask, i) {
-            uint8_t src = op->swizzle.in[i];
-            if (seen & SWS_COMP(src)) {
-                out->uop = SWS_UOP_COPY;
-                break;
-            }
-            seen |= SWS_COMP(src);
-        }
-        break;
-    }
-    case SWS_OP_UNPACK:     out->uop = SWS_UOP_UNPACK;     break;
-    case SWS_OP_PACK:       out->uop = SWS_UOP_PACK;       break;
-    case SWS_OP_LSHIFT:     out->uop = SWS_UOP_LSHIFT;     break;
-    case SWS_OP_RSHIFT:     out->uop = SWS_UOP_RSHIFT;     break;
-    case SWS_OP_CLEAR:      out->uop = SWS_UOP_CLEAR;      break;
-    case SWS_OP_CONVERT:
-        if (op->convert.expand) {
-            switch (op->convert.to) {
-            case SWS_PIXEL_U16: out->uop = SWS_UOP_EXPAND_PAIR; break;
-            case SWS_PIXEL_U32: out->uop = SWS_UOP_EXPAND_QUAD; break;
-            }
-        } else {
-            switch (op->convert.to) {
-            case SWS_PIXEL_U8:  out->uop = SWS_UOP_TO_U8;  break;
-            case SWS_PIXEL_U16: out->uop = SWS_UOP_TO_U16; break;
-            case SWS_PIXEL_U32: out->uop = SWS_UOP_TO_U32; break;
-            case SWS_PIXEL_F32: out->uop = SWS_UOP_TO_F32; break;
-            }
-        }
-        break;
-    case SWS_OP_MIN:
-    case SWS_OP_MAX:
-        out->uop = (op->op == SWS_OP_MIN) ? SWS_UOP_MIN : SWS_UOP_MAX;
-        out->mask &= ff_sws_comp_mask_q4(op->clamp.limit);
-        break;
-    case SWS_OP_SCALE:      out->uop = SWS_UOP_SCALE;      break;
-    case SWS_OP_LINEAR:
-        out->uop = (ctx->flags & SWS_BITEXACT)
-                 ? SWS_UOP_LINEAR
-                 : SWS_UOP_LINEAR_FMA;
-        break;
-    case SWS_OP_DITHER:     out->uop = SWS_UOP_DITHER;     break;
-    default:
-        return AVERROR(ENOTSUP);
-    }
-
     switch (out->uop) {
-    case SWS_UOP_READ_BIT:
-    case SWS_UOP_READ_NIBBLE:
-    case SWS_UOP_READ_PACKED:
-    case SWS_UOP_READ_PLANAR:
-    case SWS_UOP_WRITE_BIT:
-    case SWS_UOP_WRITE_NIBBLE:
-    case SWS_UOP_WRITE_PACKED:
-    case SWS_UOP_WRITE_PLANAR:
-        switch (op->rw.elems) {
-        case 1: out->mask = SWS_COMP_ELEMS(1); break;
-        case 2: out->mask = SWS_COMP_ELEMS(2); break;
-        case 3: out->mask = SWS_COMP_ELEMS(3); break;
-        case 4: out->mask = SWS_COMP_ELEMS(4); break;
-        };
-        break;
     case SWS_UOP_PERMUTE:
-    case SWS_UOP_COPY:
+    case SWS_UOP_COPY: {
         /* Recompute mask taking identity swizzle into account */
         out->mask = 0;
-        for (int i = 0; i < 4; i++) {
-            if (SWS_OP_NEEDED(op, i) && op->swizzle.in[i] != i)
-                out->mask |= SWS_COMP(i);
+        for (int i = 0; i < out->par.move.num_moves; i++) {
+            int dst = out->par.move.dst[i];
+            if (dst >= 0)
+                out->mask |= SWS_COMP(dst);
         }
-        convert_swizzle_to_moves(op, out);
+
         /* The element size and type don't matter. */
-        out->block_size = block_size * ff_sws_pixel_type_size(op->type);
+        out->block_size = block_size * ff_sws_pixel_type_size(out->type);
         out->type = SWS_PIXEL_U8;
         break;
-    case SWS_UOP_UNPACK:
-        for (int i = 0; i < 4; i++)
-            out->par.pack.pattern[i] = op->pack.pattern[i];
-        break;
-    case SWS_UOP_PACK:
-        out->mask = 0;
-        for (int i = 0; i < 4 && op->pack.pattern[i]; i++)
-            out->mask |= SWS_COMP(i);
-        for (int i = 0; i < 4; i++)
-            out->par.pack.pattern[i] = op->pack.pattern[i];
-        break;
-    case SWS_UOP_LSHIFT:
-    case SWS_UOP_RSHIFT:
-        out->par.shift.amount = op->shift.amount;
-        break;
-    case SWS_UOP_CLEAR:
-        out->mask = 0;
-        for (int i = 0; i < 4; i++) {
-            if (op->clear.mask & SWS_COMP(i)) {
-                out->mask |= SWS_COMP(i);
-                if (op->clear.value[i].num == 0) {
-                    out->par.clear.zero |= SWS_COMP(i);
-                } else {
-                    uint32_t val = op->clear.value[i].num / op->clear.value[i].den;
-                    if ((op->type == SWS_PIXEL_U8  && val == UINT8_MAX)  ||
-                        (op->type == SWS_PIXEL_U16 && val == UINT16_MAX) ||
-                        (op->type == SWS_PIXEL_U32 && val == UINT32_MAX))
-                        out->par.clear.one |= SWS_COMP(i);
-                }
-            }
-        }
-        break;
-    case SWS_UOP_LINEAR:
+    }
     case SWS_UOP_LINEAR_FMA:
-        out->mask = 0;
-        const uint32_t lin_mask = ff_sws_linear_mask(&op->lin);
-        for (int i = 0; i < 4; i++) {
-            if (!SWS_OP_NEEDED(op, i) || !(lin_mask & SWS_MASK_ROW(i))) {
-                for (int j = 0; j < 5; j++)
-                    out->par.lin.zero |= SWS_MASK(i, j);
-                continue;
-            }
-            out->mask |= SWS_COMP(i);
-            for (int j = 0; j < 5; j++) {
-                const AVRational64 k = op->lin.m[i][j];
-                if (j < 4 && k.num == k.den)
-                    out->par.lin.one |= SWS_MASK(i, j);
-                else if (k.num == 0)
-                    out->par.lin.zero |= SWS_MASK(i, j);
-            }
-        }
+        /* par.lin.exact is currently unused by asmgen_op_linear(). */
+        out->par.lin.exact = 0;
         break;
-    case SWS_UOP_DITHER:
-        out->mask = SWS_COMP_MASK(op->dither.y_offset[0] >= 0,
-                                  op->dither.y_offset[1] >= 0,
-                                  op->dither.y_offset[2] >= 0,
-                                  op->dither.y_offset[3] >= 0);
-        LOOP(out->mask, i) {
-            out->par.dither.y_offset[i] = op->dither.y_offset[i];
-        }
-        out->par.dither.size_log2 = op->dither.size_log2;
-        break;
-    }
-
-    switch (out->uop) {
-    case SWS_UOP_READ_BIT:
-    case SWS_UOP_READ_NIBBLE:
-    case SWS_UOP_READ_PACKED:
-    case SWS_UOP_READ_PLANAR:
-    case SWS_UOP_WRITE_BIT:
-    case SWS_UOP_WRITE_NIBBLE:
-    case SWS_UOP_WRITE_PACKED:
-    case SWS_UOP_WRITE_PLANAR:
-    case SWS_UOP_SWAP_BYTES:
-    case SWS_UOP_CLEAR:
-        /* Only the element size matters, not the type. */
-        if (out->type == SWS_PIXEL_F32)
-            out->type = SWS_PIXEL_U32;
+    default:
         break;
     }
-
-    return 0;
 }
diff --git a/libswscale/tests/sws_ops_aarch64.c b/libswscale/tests/sws_ops_aarch64.c
index 2155319a33..6ce1ab67f4 100644
--- a/libswscale/tests/sws_ops_aarch64.c
+++ b/libswscale/tests/sws_ops_aarch64.c
@@ -197,16 +197,25 @@ static int collect_ops_compile(SwsContext *ctx, const SwsOpList *ops,
     struct AVTreeNode **root = (struct AVTreeNode **) ctx->opaque;
     int ret;

+    SwsUOpList *uops = ff_sws_uop_list_alloc();
+    if (!uops)
+        return AVERROR(ENOMEM);
+
+    const SwsUOpFlags flags = (ctx->flags & SWS_BITEXACT) ? 0 : SWS_UOP_FLAG_FMA;
+    ret = ff_sws_ops_translate(ctx, ops, flags, uops);
+    if (ret == AVERROR(ENOTSUP)) {
+        ret = 0;
+        goto end;
+    }
+    if (ret < 0)
+        goto end;
+
     /* Use at most two full vregs during the widest precision section */
-    int block_size = (ff_sws_op_list_max_size(ops) == 4) ? 8 : 16;
+    int block_size = (uops->pixel_size_max == 4) ? 8 : 16;

-    for (int i = 0; i < ops->num_ops; i++) {
+    for (int i = 0; i < uops->num_ops; i++) {
         SwsAArch64OpImplParams params = { 0 };
-        ret = convert_to_aarch64_impl(ctx, ops, i, block_size, &params);
-        if (ret == AVERROR(ENOTSUP))
-            continue;
-        if (ret < 0)
-            goto end;
+        convert_to_aarch64_impl(&uops->ops[i], block_size, &params);
         ret = aarch64_collect_op(&params, root);
         if (ret < 0)
             goto end;
@@ -226,6 +235,7 @@ static int collect_ops_compile(SwsContext *ctx, const SwsOpList *ops,
     ret = 0;

 end:
+    ff_sws_uop_list_free(&uops);
     return ret;
 }