Commit d1fd9c7b9c for ffmpeg
commit d1fd9c7b9cc5fc3c165115264e2852ac6e075906
Author: Ramiro Polla <ramiro.polla@gmail.com>
Date: Sun Jul 19 13:22:54 2026 +0200
swscale/aarch64/ops_asmgen: implement integer linear transformations
checkasm:
- CPU: ARM Cortex-A720
- Timing source: linux (perf)
Benchmark results:
name ticks (vs ref)
u8_linear_x_x000x_c: 1117.8
u8_linear_x_x000x_aarch64_neon: 463.6 ( 2.41x)
u8_linear_xyz_x000x_0x00x_00x0x_c: 1269.2
u8_linear_xyz_x000x_0x00x_00x0x_aarch64_neon: 586.3 ( 2.16x)
u8_linear_xyz_x0000_0x000_00x00_c: 1270.5
u8_linear_xyz_x0000_0x000_00x00_aarch64_neon: 503.3 ( 2.52x)
u8_linear_y_0x000_c: 1274.3
u8_linear_y_0x000_aarch64_neon: 443.8 ( 2.87x)
u16_linear_x_x000x_c: 1627.6
u16_linear_x_x000x_aarch64_neon: 565.8 ( 2.88x)
u16_linear_xyz_x000x_0x00x_00x0x_c: 1807.0
u16_linear_xyz_x000x_0x00x_00x0x_aarch64_neon: 830.5 ( 2.18x)
u16_linear_xyz_x0000_0x000_00x00_c: 1686.4
u16_linear_xyz_x0000_0x000_00x00_aarch64_neon: 585.9 ( 2.88x)
u16_linear_xyzw_x0000_0x000_00x00_000x0_c: 1753.8
u16_linear_xyzw_x0000_0x000_00x00_000x0_aarch64_neon: 680.4 ( 2.58x)
u32_linear_x_x000x_c: 2617.0
u32_linear_x_x000x_aarch64_neon: 1007.5 ( 2.60x)
u32_linear_xyz_x000x_0x00x_00x0x_c: 3107.7
u32_linear_xyz_x000x_0x00x_00x0x_aarch64_neon: 1504.7 ( 2.06x)
u32_linear_xyz_x0000_0x000_00x00_c: 3106.0
u32_linear_xyz_x0000_0x000_00x00_aarch64_neon: 1008.3 ( 3.08x)
On in-order cores (Cortex-A520), checkasm reports NEON slower than C for
most op chains. This is mainly an artifact of the checkasm buffer layout:
the input and output planes are 16 KiB apart and conflict in the 4-way
32 KiB L1D, and the smaller NEON block sizes (8/16 vs 32) touch each
cache line more often. In swscale benchmarks on the same core, the
conversions that now use integer linear ops are 1.1x-2.9x faster than
with the previous f32 NEON path.
Sponsored-by: Sovereign Tech Fund
Signed-off-by: Ramiro Polla <ramiro.polla@gmail.com>
diff --git a/libswscale/aarch64/ops.c b/libswscale/aarch64/ops.c
index b02312b4b7..f1d3d698ab 100644
--- a/libswscale/aarch64/ops.c
+++ b/libswscale/aarch64/ops.c
@@ -74,7 +74,7 @@ static int aarch64_setup_linear(const SwsAArch64OpImplParams *p,
*/
const int num_vregs = linear_num_vregs(p);
av_assert0(num_vregs <= 4);
- float *coeffs = av_malloc(num_vregs * 4 * sizeof(float));
+ void *coeffs = av_malloc(num_vregs * 16);
if (!coeffs)
return AVERROR(ENOMEM);
@@ -86,8 +86,15 @@ static int aarch64_setup_linear(const SwsAArch64OpImplParams *p,
for (int i = 0; i < 4; i++) {
for (int j = 0; j < 5; j++) {
const int jj = (j == 0) ? 4 : (j - 1);
- if (!(p->par.lin.zero & SWS_MASK(i, jj)))
- coeffs[i_coeff++] = uop->data.mat4[i][jj].f32;
+ if (!(p->par.lin.zero & SWS_MASK(i, jj))) {
+ const SwsPixel px = uop->data.mat4[i][jj];
+ switch (p->type) {
+ case SWS_PIXEL_U8: ((uint8_t *) coeffs)[i_coeff++] = px.u8; break;
+ case SWS_PIXEL_U16: ((uint16_t *) coeffs)[i_coeff++] = px.u16; break;
+ case SWS_PIXEL_U32: ((uint32_t *) coeffs)[i_coeff++] = px.u32; break;
+ case SWS_PIXEL_F32: ((float *) coeffs)[i_coeff++] = px.f32; break;
+ }
+ }
}
}
diff --git a/libswscale/aarch64/ops_asmgen.c b/libswscale/aarch64/ops_asmgen.c
index f83e487bc7..b2eb0b178d 100644
--- a/libswscale/aarch64/ops_asmgen.c
+++ b/libswscale/aarch64/ops_asmgen.c
@@ -875,9 +875,14 @@ static void linear_pass(SwsAArch64Context *s, const SwsAArch64OpImplParams *p,
} else if (first && !is_offset) {
if (p->par.lin.one & SWS_MASK(i, src_j)) {
i_mov16b(r, dx[i], vsrc); CMTF("v%c[%u] = vsrc%c[%u];", cvh, i, cvh, src_j);
- } else {
+ } else if (p->type == SWS_PIXEL_F32) {
i_fmul (r, dx[i], vsrc, vcoeff); CMTF("v%c[%u] = vsrc%c[%u] * coeff[%u][%u];", cvh, i, cvh, src_j, i, src_j);
+ } else {
+ i_mul (r, dx[i], vsrc, vcoeff); CMTF("v%c[%u] = vsrc%c[%u] * coeff[%u][%u];", cvh, i, cvh, src_j, i, src_j);
}
+ } else if (p->type != SWS_PIXEL_F32) {
+ /* Integer multiply-accumulate is always exact. */
+ i_mla (r, dx[i], vsrc, vcoeff); CMTF("v%c[%u] += vsrc%c[%u] * coeff[%u][%u];", cvh, i, cvh, src_j, i, src_j);
} else if (p->uop == SWS_UOP_LINEAR_FMA) {
/**
* Most modern aarch64 cores have a fastpath for sequences
diff --git a/libswscale/aarch64/ops_impl.h b/libswscale/aarch64/ops_impl.h
index 560021e4ce..c9ad019cfe 100644
--- a/libswscale/aarch64/ops_impl.h
+++ b/libswscale/aarch64/ops_impl.h
@@ -63,6 +63,12 @@ typedef struct SwsAArch64OpImplParams {
#define LOOP_MASK(p, idx) LOOP(p->mask, idx)
#define LOOP_MASK_BWD(p, idx) LOOP_BWD(p->mask, idx)
+/* Number of elements in each linear coefficient vector register. */
+static inline int linear_vreg_nelems(SwsPixelType type)
+{
+ return 16 / ff_sws_pixel_type_size(type);
+}
+
/* Compute number of vector registers needed to store all coefficients. */
static inline int linear_num_vregs(const SwsAArch64OpImplParams *params)
{
@@ -70,7 +76,8 @@ static inline int linear_num_vregs(const SwsAArch64OpImplParams *params)
for (int i = 0; i < 4 * 5; i++)
if (!(params->par.lin.zero & (1ULL << i)))
count++;
- return (count + 3) / 4;
+ const int nelems = linear_vreg_nelems(params->type);
+ return (count + nelems - 1) / nelems;
}
/**
diff --git a/libswscale/aarch64/ops_static.c b/libswscale/aarch64/ops_static.c
index 74891d2be3..2a7933d675 100644
--- a/libswscale/aarch64/ops_static.c
+++ b/libswscale/aarch64/ops_static.c
@@ -307,6 +307,19 @@ static void asmgen_setup_linear(SwsAArch64Context *s, const SwsAArch64OpImplPara
asmgen_set_load_cont_node(s);
i_ld1(r, coeff_veclist, a64op_base(ptr)); CMT("coeff_veclist = *vcoeff_ptr;");
+ /**
+ * Integer mul/mla by element does not exist for 8-bit elements, and
+ * requires the element register to be in v0-v15 for 16-bit elements.
+ * For these types, the coefficients used as multiplication operands
+ * are broadcast into full temp vectors instead. Temp vectors 8-11
+ * are otherwise only used by the floating-point fmul+fadd path, so
+ * they are free for integer types.
+ */
+ const bool dup_coeffs = (p->type == SWS_PIXEL_U8 || p->type == SWS_PIXEL_U16);
+ const int nelems = linear_vreg_nelems(p->type);
+ RasmOp *vint = &vt[8];
+ int i_vint = 0;
+
/**
* Populate operands matrix from packed data into linear_vcoeff matrix
* and compute mask for rows that must be saved before being overwritten.
@@ -315,15 +328,28 @@ static void asmgen_setup_linear(SwsAArch64Context *s, const SwsAArch64OpImplPara
bool overwritten[4] = { false, false, false, false };
int i_coeff = 0;
LOOP_MASK(p, i) {
+ bool first = true;
for (int j = 0; j < 5; j++) {
bool is_offset = (j == 0);
int src_j = is_offset ? 4 : (j - 1);
if (p->par.lin.zero & SWS_MASK(i, src_j))
continue;
- uint8_t vc_i = i_coeff / 4;
- uint8_t vc_j = i_coeff & 3;
- regs->linear_vcoeff[i][j] = a64op_elem(vc[vc_i], vc_j);
+ uint8_t vc_i = i_coeff / nelems;
+ uint8_t vc_j = i_coeff & (nelems - 1);
+ RasmOp vcoeff = a64op_elem(vc[vc_i], vc_j);
+ /**
+ * The first coefficient of a row is not used as a multiplication
+ * operand if it is either an offset (broadcast) or one (move).
+ */
+ bool is_one = !is_offset && (p->par.lin.one & SWS_MASK(i, src_j));
+ if (dup_coeffs && !(first && (is_offset || is_one))) {
+ av_assert0(i_vint < 4);
+ i_dup(r, vint[i_vint], vcoeff); CMTF("vcoeff%u = broadcast(coeff[%u][%u]);", i_vint, i, src_j);
+ vcoeff = vint[i_vint++];
+ }
+ regs->linear_vcoeff[i][j] = vcoeff;
i_coeff++;
+ first = false;
if (!is_offset && overwritten[src_j])
save_mask |= SWS_COMP(src_j);
overwritten[i] = true;
diff --git a/libswscale/tests/sws_ops_aarch64.c b/libswscale/tests/sws_ops_aarch64.c
index 6ce1ab67f4..05eba0f0bb 100644
--- a/libswscale/tests/sws_ops_aarch64.c
+++ b/libswscale/tests/sws_ops_aarch64.c
@@ -219,7 +219,7 @@ static int collect_ops_compile(SwsContext *ctx, const SwsOpList *ops,
ret = aarch64_collect_op(¶ms, root);
if (ret < 0)
goto end;
- if (params.uop == SWS_UOP_LINEAR_FMA) {
+ if (params.uop == SWS_UOP_LINEAR_FMA && params.type == SWS_PIXEL_F32) {
/**
* Generate both sets of linear op functions that do use
* and do not use fmla (selected by SWS_BITEXACT).