Commit 00d08d1955 for ffmpeg
commit 00d08d1955f036571830909d0af6c34872f4bfe6
Author: Niklas Haas <git@haasn.dev>
Date: Fri Jul 17 12:51:54 2026 +0200
swscale/x86/ops_int: implement integer linear transformations
This is a bit inefficient for the hyper-special case of e.g. a single
integer multiplication only. In particular, the AVX2 path actually ends up
slower than the SSE4 path; though we still need to implement it due to the
block size needing to match the rest of the chain.
I plan on maybe adding a special case to cover the isolated / single
component case down the line.
checkasm:
- CPU: AMD Ryzen 9 9950X3D 16-Core Processor (00B40F40)
- Timing source: x86 (rdtsc)
Benchmark results:
name cycles (vs ref)
u8_linear_x_x000x_c: 1039.5
u8_linear_x_x000x_x86_sse4: 428.5 ( 2.41x)
u8_linear_x_x000x_x86_avx2: 907.4 ( 1.15x)
u8_linear_xyz_x000x_0x00x_00x0x_c: 1177.7
u8_linear_xyz_x000x_0x00x_00x0x_x86_sse4: 714.1 ( 1.65x)
u8_linear_xyz_x000x_0x00x_00x0x_x86_avx2: 843.1 ( 1.39x)
u8_linear_xyz_x0000_0x000_00x00_c: 1131.5
u8_linear_xyz_x0000_0x000_00x00_x86_sse4: 713.6 ( 1.58x)
u8_linear_xyz_x0000_0x000_00x00_x86_avx2: 867.3 ( 1.30x)
u8_linear_y_0x000_c: 1031.3
u8_linear_y_0x000_x86_sse4: 389.2 ( 2.65x)
u8_linear_y_0x000_x86_avx2: 911.5 ( 1.13x)
u16_linear_x_x000x_c: 1206.4
u16_linear_x_x000x_x86_avx2: 610.4 ( 1.98x)
u16_linear_xyz_x000x_0x00x_00x0x_c: 1445.1
u16_linear_xyz_x000x_0x00x_00x0x_x86_avx2: 503.6 ( 2.87x)
u16_linear_xyz_x0000_0x000_00x00_c: 1324.8
u16_linear_xyz_x0000_0x000_00x00_x86_avx2: 506.2 ( 2.62x)
u16_linear_xyzw_x0000_0x000_00x00_000x0_c: 1424.6
u16_linear_xyzw_x0000_0x000_00x00_000x0_x86_avx2: 467.0 ( 3.05x)
u32_linear_x_x000x_c: 1762.4
u32_linear_x_x000x_x86_avx2: 855.9 ( 2.06x)
u32_linear_xyz_x000x_0x00x_00x0x_c: 2461.8
u32_linear_xyz_x000x_0x00x_00x0x_x86_avx2: 919.9 ( 2.68x)
u32_linear_xyz_x0000_0x000_00x00_c: 2297.8
u32_linear_xyz_x0000_0x000_00x00_x86_avx2: 865.6 ( 2.65x)
Sponsored-by: Sovereign Tech Fund
Signed-off-by: Niklas Haas <git@haasn.dev>
diff --git a/libswscale/x86/ops.c b/libswscale/x86/ops.c
index 00994e4d64..e79b1cf117 100644
--- a/libswscale/x86/ops.c
+++ b/libswscale/x86/ops.c
@@ -277,12 +277,52 @@ static int setup_dither(const SwsImplParams *params, SwsImplResult *out)
return 0;
}
+static void splat_lane(void *dst, SwsPixelType type, SwsPixel px)
+{
+ switch (ff_sws_pixel_type_size(type)) {
+ case 1:
+ memset(dst, px.u8, 16);
+ break;
+ case 2:
+ for (int i = 0; i < 8; i++)
+ ((uint16_t *) dst)[i] = px.u16;
+ break;
+ case 4:
+ for (int i = 0; i < 4; i++)
+ ((uint32_t *) dst)[i] = px.u32;
+ break;
+ }
+}
+
static int setup_linear(const SwsImplParams *params, SwsImplResult *out)
{
const SwsUOp *uop = params->uop;
- out->priv.ptr = av_memdup(uop->data.mat4, sizeof(uop->data.mat4));
+ if (uop->type == SWS_PIXEL_F32) {
+ out->priv.ptr = av_memdup(uop->data.mat4, sizeof(uop->data.mat4));
+ out->free = ff_op_priv_free;
+ return out->priv.ptr ? 0 : AVERROR(ENOMEM);
+ }
+
+ uint8_t *mat = av_malloc(4 * 5 * 16); /* one lane per component */
+ if (!mat)
+ return AVERROR(ENOMEM);
+ out->priv.ptr = mat;
out->free = ff_op_priv_free;
- return out->priv.ptr ? 0 : AVERROR(ENOMEM);
+
+ for (int i = 0; i < 4; i++) {
+ for (int j = 0; j < 5; j++) {
+ SwsPixel px = uop->data.mat4[i][j];
+ SwsPixelType type = uop->type;
+ if (type == SWS_PIXEL_U8) {
+ type = SWS_PIXEL_U16; /* for pmullw */
+ px.u16 = (px.u8 << 8) | px.u8;
+ }
+
+ splat_lane(mat, type, px);
+ mat += 16;
+ }
+ }
+ return 0;
}
static bool uop_is_type_invariant(const SwsUOpType uop)
diff --git a/libswscale/x86/ops_float.asm b/libswscale/x86/ops_float.asm
index 3b8055c31a..2c1858c14a 100644
--- a/libswscale/x86/ops_float.asm
+++ b/libswscale/x86/ops_float.asm
@@ -534,9 +534,7 @@ IF W, maxps mw2, m11
;---------------------------------------------------------
; Linear operations
-%define LIN_MASK(I, J) (1 << (5 * (I) + (J)))
-
-%macro linear_muladd 5 ; dst, src, use_coef, coef, use_fma
+%macro linear_muladdps 5 ; dst, src, use_coef, coef, use_fma
%if INIT ; dst is already initialized
%if %3 && %5
fmaddps %1, %4, %2, %1
@@ -570,10 +568,10 @@ IF LOAD(0), vbroadcastss m12, [%2 + 0 * BYTES]
IF LOAD(1), vbroadcastss m13, [%2 + 1 * BYTES]
IF LOAD(2), vbroadcastss m14, [%2 + 2 * BYTES]
IF LOAD(3), vbroadcastss m15, [%2 + 3 * BYTES]
-IF NEED(0), linear_muladd %1, mx%4, LOAD(0), m12, FMA(0)
-IF NEED(1), linear_muladd %1, my%4, LOAD(1), m13, FMA(1)
-IF NEED(2), linear_muladd %1, mz%4, LOAD(2), m14, FMA(2)
-IF NEED(3), linear_muladd %1, mw%4, LOAD(3), m15, FMA(3)
+IF NEED(0), linear_muladdps %1, mx%4, LOAD(0), m12, FMA(0)
+IF NEED(1), linear_muladdps %1, my%4, LOAD(1), m13, FMA(1)
+IF NEED(2), linear_muladdps %1, mz%4, LOAD(2), m14, FMA(2)
+IF NEED(3), linear_muladdps %1, mw%4, LOAD(3), m15, FMA(3)
assert INIT, SWS_UOP_LINEAR should not contain empty rows
%endmacro
diff --git a/libswscale/x86/ops_include.asm b/libswscale/x86/ops_include.asm
index 073ed31e57..85777d7529 100644
--- a/libswscale/x86/ops_include.asm
+++ b/libswscale/x86/ops_include.asm
@@ -146,6 +146,9 @@ endstruc
%define SWS_COMP_INV(mask) ((mask) ^ SWS_COMP_ALL)
%define SWS_COMP_ELEMS(N) ((1 << (N)) - 1)
+%define LIN_MASK(I, J) (1 << (5 * (I) + (J)))
+%define LIN_COL(J) (LIN_MASK(0, J) | LIN_MASK(1, J) | LIN_MASK(2, J) | LIN_MASK(3, J))
+
;---------------------------------------------------------
; Common macros for declaring operations
@@ -326,13 +329,19 @@ endstruc
%endif
%endmacro
-; Alternate name; for nested usage (to work around NASM limitations)
+; Alternate names; for nested usage (to work around NASM limitations)
%macro IF1 2+
%if %1
%2
%endif
%endmacro
+%macro IF2 2+
+ %if %1
+ %2
+ %endif
+%endmacro
+
%macro shl_log2 2 ; dst, amount
%if %2 == 64
shl %1, 6
diff --git a/libswscale/x86/ops_int.asm b/libswscale/x86/ops_int.asm
index 76e369d1fa..6f49ffeafe 100644
--- a/libswscale/x86/ops_int.asm
+++ b/libswscale/x86/ops_int.asm
@@ -770,6 +770,113 @@ assert 0, SWS_UOP_LINEAR_FMA is not implemented for integer types
assert 0, SWS_UOP_DITHER is not implemented for integer types
%endmacro
+;---------------------------------------------------------
+; Linear operations
+
+%macro linear_muladdw 4 ; dst, src, use_coef, coef
+ %if BITS == 32
+ %xdefine MUL pmulld
+ %xdefine ADD paddd
+ %else
+ %xdefine MUL pmullw
+ %xdefine ADD paddw
+ %endif
+ %if INIT ; dst is already initialized
+ %if %3
+ MUL %4, %2
+ ADD %1, %4
+ %else
+ ADD %1, %2
+ %endif
+ %else
+ %assign INIT 1
+ %if %3
+ MUL %1, %2, %4
+ %else
+ mova %1, %2
+ %endif
+ %endif
+%endmacro
+
+%macro linear_row 3 ; dst, src, row
+%xdefine NEED(J) (!(ZERO_MASK & LIN_MASK(%3, J)))
+%xdefine LOAD(J) (NEED(J) && !(ONE_MASK & LIN_MASK(%3, J)))
+%assign INIT 0 ; track whether `dst` already contains data
+
+ %if !(ZERO_MASK & LIN_MASK(%3, 4)) ; nonzero output offset
+ %assign INIT 1
+ VBROADCASTI128 %1, [%2 + 4 * 16]
+ %endif
+IF LOAD(0), VBROADCASTI128 m12, [%2 + 0 * 16]
+IF LOAD(1), VBROADCASTI128 m13, [%2 + 1 * 16]
+IF LOAD(2), VBROADCASTI128 m14, [%2 + 2 * 16]
+IF LOAD(3), VBROADCASTI128 m15, [%2 + 3 * 16]
+IF NEED(0), linear_muladdw %1, IN0, LOAD(0), m12
+IF NEED(1), linear_muladdw %1, IN1, LOAD(1), m13
+IF NEED(2), linear_muladdw %1, IN2, LOAD(2), m14
+IF NEED(3), linear_muladdw %1, IN3, LOAD(3), m15
+ assert INIT, SWS_UOP_LINEAR should not contain empty rows
+%endmacro
+
+; Swap the high and low bytes of `dst` and `out` and merge back into `dst`
+%macro linear_rot 4 ; have_out, need_in, dst, out
+ %if %1 || %2 ; we also need to rotate pure input registers
+ %if %1
+ psllw %4, 8
+ %else
+ psllw %4, %3, 8
+ %endif
+ psrlw %3, 8
+ por %3, %4
+ %endif
+%endmacro
+
+%macro linear_pass 0-1 ; suffix
+%xdefine USED(J) (!(ZERO_MASK & LIN_COL(J)))
+%xdefine IN0 mx%1
+%xdefine IN1 my%1
+%xdefine IN2 mz%1
+%xdefine IN3 mw%1
+
+IF1 X, linear_row m8, tmp0q + 0 * 16, 0
+IF1 Y, linear_row m9, tmp0q + 5 * 16, 1
+IF1 Z, linear_row m10, tmp0q + 10 * 16, 2
+IF1 W, linear_row m11, tmp0q + 15 * 16, 3
+
+ %if BITS == 8
+ ; swap high/low bits and compute the other half; this discards the
+ ; garbage high byte produced by each sub-pass
+ linear_rot X, USED(0), IN0, m8
+ linear_rot Y, USED(1), IN1, m9
+ linear_rot Z, USED(2), IN2, m10
+ linear_rot W, USED(3), IN3, m11
+IF1 X, linear_row m8, tmp0q + 0 * 16, 0
+IF1 Y, linear_row m9, tmp0q + 5 * 16, 1
+IF1 Z, linear_row m10, tmp0q + 10 * 16, 2
+IF1 W, linear_row m11, tmp0q + 15 * 16, 3
+ linear_rot X, USED(0), IN0, m8
+ linear_rot Y, USED(1), IN1, m9
+ linear_rot Z, USED(2), IN2, m10
+ linear_rot W, USED(3), IN3, m11
+ %else
+IF X, mova IN0, m8
+IF Y, mova IN1, m9
+IF Z, mova IN2, m10
+IF W, mova IN3, m11
+ %endif
+%endmacro
+
+%macro LINEAR 2
+%assign ONE_MASK %1
+%assign ZERO_MASK %2
+
+ mov tmp0q, [implq + SwsOpImpl.priv] ; address of matrix
+ LOAD_CONT tmp1q
+ linear_pass
+IF2 V2, linear_pass 2
+ CONTINUE tmp1q
+%endmacro
+
;---------------------------------------------------------
; Instantiate above macros to generate all uop kernels
@@ -795,6 +902,7 @@ assert 0, SWS_UOP_DITHER is not implemented for integer types
DECL_%1_RSHIFT (RSHIFT)
DECL_%1_LINEAR_FMA (LINEAR_FMA)
DECL_%1_DITHER (DITHER)
+ DECL_%1_LINEAR (LINEAR)
%endmacro
%macro decl_type_invariant 0