Commit 00d08d1955 for ffmpeg

commit 00d08d1955f036571830909d0af6c34872f4bfe6
Author: Niklas Haas <git@haasn.dev>
Date:   Fri Jul 17 12:51:54 2026 +0200

    swscale/x86/ops_int: implement integer linear transformations

    This is a bit inefficient for the hyper-special case of e.g. a single
    integer multiplication only. In particular, the AVX2 path actually ends up
    slower than the SSE4 path; though we still need to implement it due to the
    block size needing to match the rest of the chain.

    I plan on maybe adding a special case to cover the isolated / single
    component case down the line.

    checkasm:
     - CPU: AMD Ryzen 9 9950X3D 16-Core Processor (00B40F40)
     - Timing source: x86 (rdtsc)
    Benchmark results:
      name                                                cycles (vs ref)
      u8_linear_x_x000x_c:                                1039.5
      u8_linear_x_x000x_x86_sse4:                          428.5 ( 2.41x)
      u8_linear_x_x000x_x86_avx2:                          907.4 ( 1.15x)
      u8_linear_xyz_x000x_0x00x_00x0x_c:                  1177.7
      u8_linear_xyz_x000x_0x00x_00x0x_x86_sse4:            714.1 ( 1.65x)
      u8_linear_xyz_x000x_0x00x_00x0x_x86_avx2:            843.1 ( 1.39x)
      u8_linear_xyz_x0000_0x000_00x00_c:                  1131.5
      u8_linear_xyz_x0000_0x000_00x00_x86_sse4:            713.6 ( 1.58x)
      u8_linear_xyz_x0000_0x000_00x00_x86_avx2:            867.3 ( 1.30x)
      u8_linear_y_0x000_c:                                1031.3
      u8_linear_y_0x000_x86_sse4:                          389.2 ( 2.65x)
      u8_linear_y_0x000_x86_avx2:                          911.5 ( 1.13x)
      u16_linear_x_x000x_c:                               1206.4
      u16_linear_x_x000x_x86_avx2:                         610.4 ( 1.98x)
      u16_linear_xyz_x000x_0x00x_00x0x_c:                 1445.1
      u16_linear_xyz_x000x_0x00x_00x0x_x86_avx2:           503.6 ( 2.87x)
      u16_linear_xyz_x0000_0x000_00x00_c:                 1324.8
      u16_linear_xyz_x0000_0x000_00x00_x86_avx2:           506.2 ( 2.62x)
      u16_linear_xyzw_x0000_0x000_00x00_000x0_c:          1424.6
      u16_linear_xyzw_x0000_0x000_00x00_000x0_x86_avx2:    467.0 ( 3.05x)
      u32_linear_x_x000x_c:                               1762.4
      u32_linear_x_x000x_x86_avx2:                         855.9 ( 2.06x)
      u32_linear_xyz_x000x_0x00x_00x0x_c:                 2461.8
      u32_linear_xyz_x000x_0x00x_00x0x_x86_avx2:           919.9 ( 2.68x)
      u32_linear_xyz_x0000_0x000_00x00_c:                 2297.8
      u32_linear_xyz_x0000_0x000_00x00_x86_avx2:           865.6 ( 2.65x)

    Sponsored-by: Sovereign Tech Fund
    Signed-off-by: Niklas Haas <git@haasn.dev>

diff --git a/libswscale/x86/ops.c b/libswscale/x86/ops.c
index 00994e4d64..e79b1cf117 100644
--- a/libswscale/x86/ops.c
+++ b/libswscale/x86/ops.c
@@ -277,12 +277,52 @@ static int setup_dither(const SwsImplParams *params, SwsImplResult *out)
     return 0;
 }

+static void splat_lane(void *dst, SwsPixelType type, SwsPixel px)
+{
+    switch (ff_sws_pixel_type_size(type)) {
+    case 1:
+        memset(dst, px.u8, 16);
+        break;
+    case 2:
+        for (int i = 0; i < 8; i++)
+            ((uint16_t *) dst)[i] = px.u16;
+        break;
+    case 4:
+        for (int i = 0; i < 4; i++)
+            ((uint32_t *) dst)[i] = px.u32;
+        break;
+    }
+}
+
 static int setup_linear(const SwsImplParams *params, SwsImplResult *out)
 {
     const SwsUOp *uop = params->uop;
-    out->priv.ptr = av_memdup(uop->data.mat4, sizeof(uop->data.mat4));
+    if (uop->type == SWS_PIXEL_F32) {
+        out->priv.ptr = av_memdup(uop->data.mat4, sizeof(uop->data.mat4));
+        out->free = ff_op_priv_free;
+        return out->priv.ptr ? 0 : AVERROR(ENOMEM);
+    }
+
+    uint8_t *mat = av_malloc(4 * 5 * 16); /* one lane per component */
+    if (!mat)
+        return AVERROR(ENOMEM);
+    out->priv.ptr = mat;
     out->free = ff_op_priv_free;
-    return out->priv.ptr ? 0 : AVERROR(ENOMEM);
+
+    for (int i = 0; i < 4; i++) {
+        for (int j = 0; j < 5; j++) {
+            SwsPixel px = uop->data.mat4[i][j];
+            SwsPixelType type = uop->type;
+            if (type == SWS_PIXEL_U8) {
+                type = SWS_PIXEL_U16; /* for pmullw */
+                px.u16 = (px.u8 << 8) | px.u8;
+            }
+
+            splat_lane(mat, type, px);
+            mat += 16;
+        }
+    }
+    return 0;
 }

 static bool uop_is_type_invariant(const SwsUOpType uop)
diff --git a/libswscale/x86/ops_float.asm b/libswscale/x86/ops_float.asm
index 3b8055c31a..2c1858c14a 100644
--- a/libswscale/x86/ops_float.asm
+++ b/libswscale/x86/ops_float.asm
@@ -534,9 +534,7 @@ IF W,   maxps mw2, m11
 ;---------------------------------------------------------
 ; Linear operations

-%define LIN_MASK(I, J) (1 << (5 * (I) + (J)))
-
-%macro linear_muladd 5 ; dst, src, use_coef, coef, use_fma
+%macro linear_muladdps 5 ; dst, src, use_coef, coef, use_fma
     %if INIT ; dst is already initialized
         %if %3 && %5
             fmaddps %1, %4, %2, %1
@@ -570,10 +568,10 @@ IF LOAD(0), vbroadcastss m12, [%2 + 0 * BYTES]
 IF LOAD(1), vbroadcastss m13, [%2 + 1 * BYTES]
 IF LOAD(2), vbroadcastss m14, [%2 + 2 * BYTES]
 IF LOAD(3), vbroadcastss m15, [%2 + 3 * BYTES]
-IF NEED(0), linear_muladd %1, mx%4, LOAD(0), m12, FMA(0)
-IF NEED(1), linear_muladd %1, my%4, LOAD(1), m13, FMA(1)
-IF NEED(2), linear_muladd %1, mz%4, LOAD(2), m14, FMA(2)
-IF NEED(3), linear_muladd %1, mw%4, LOAD(3), m15, FMA(3)
+IF NEED(0), linear_muladdps %1, mx%4, LOAD(0), m12, FMA(0)
+IF NEED(1), linear_muladdps %1, my%4, LOAD(1), m13, FMA(1)
+IF NEED(2), linear_muladdps %1, mz%4, LOAD(2), m14, FMA(2)
+IF NEED(3), linear_muladdps %1, mw%4, LOAD(3), m15, FMA(3)
             assert INIT, SWS_UOP_LINEAR should not contain empty rows
 %endmacro

diff --git a/libswscale/x86/ops_include.asm b/libswscale/x86/ops_include.asm
index 073ed31e57..85777d7529 100644
--- a/libswscale/x86/ops_include.asm
+++ b/libswscale/x86/ops_include.asm
@@ -146,6 +146,9 @@ endstruc
 %define SWS_COMP_INV(mask)      ((mask) ^ SWS_COMP_ALL)
 %define SWS_COMP_ELEMS(N)       ((1 << (N)) - 1)

+%define LIN_MASK(I, J) (1 << (5 * (I) + (J)))
+%define LIN_COL(J) (LIN_MASK(0, J) | LIN_MASK(1, J) | LIN_MASK(2, J) | LIN_MASK(3, J))
+
 ;---------------------------------------------------------
 ; Common macros for declaring operations

@@ -326,13 +329,19 @@ endstruc
     %endif
 %endmacro

-; Alternate name; for nested usage (to work around NASM limitations)
+; Alternate names; for nested usage (to work around NASM limitations)
 %macro IF1 2+
     %if %1
         %2
     %endif
 %endmacro

+%macro IF2 2+
+    %if %1
+        %2
+    %endif
+%endmacro
+
 %macro shl_log2 2 ; dst, amount
     %if %2 == 64
         shl %1, 6
diff --git a/libswscale/x86/ops_int.asm b/libswscale/x86/ops_int.asm
index 76e369d1fa..6f49ffeafe 100644
--- a/libswscale/x86/ops_int.asm
+++ b/libswscale/x86/ops_int.asm
@@ -770,6 +770,113 @@ assert 0, SWS_UOP_LINEAR_FMA is not implemented for integer types
 assert 0, SWS_UOP_DITHER is not implemented for integer types
 %endmacro

+;---------------------------------------------------------
+; Linear operations
+
+%macro linear_muladdw 4 ; dst, src, use_coef, coef
+    %if BITS == 32
+        %xdefine MUL pmulld
+        %xdefine ADD paddd
+    %else
+        %xdefine MUL pmullw
+        %xdefine ADD paddw
+    %endif
+    %if INIT ; dst is already initialized
+        %if %3
+            MUL %4, %2
+            ADD %1, %4
+        %else
+            ADD %1, %2
+        %endif
+    %else
+        %assign INIT 1
+        %if %3
+            MUL %1, %2, %4
+        %else
+            mova %1, %2
+        %endif
+    %endif
+%endmacro
+
+%macro linear_row 3 ; dst, src, row
+%xdefine NEED(J) (!(ZERO_MASK & LIN_MASK(%3, J)))
+%xdefine LOAD(J) (NEED(J) && !(ONE_MASK & LIN_MASK(%3, J)))
+%assign INIT 0 ; track whether `dst` already contains data
+
+    %if !(ZERO_MASK & LIN_MASK(%3, 4)) ; nonzero output offset
+            %assign INIT 1
+            VBROADCASTI128 %1, [%2 + 4 * 16]
+    %endif
+IF LOAD(0), VBROADCASTI128 m12, [%2 + 0 * 16]
+IF LOAD(1), VBROADCASTI128 m13, [%2 + 1 * 16]
+IF LOAD(2), VBROADCASTI128 m14, [%2 + 2 * 16]
+IF LOAD(3), VBROADCASTI128 m15, [%2 + 3 * 16]
+IF NEED(0), linear_muladdw %1, IN0, LOAD(0), m12
+IF NEED(1), linear_muladdw %1, IN1, LOAD(1), m13
+IF NEED(2), linear_muladdw %1, IN2, LOAD(2), m14
+IF NEED(3), linear_muladdw %1, IN3, LOAD(3), m15
+            assert INIT, SWS_UOP_LINEAR should not contain empty rows
+%endmacro
+
+; Swap the high and low bytes of `dst` and `out` and merge back into `dst`
+%macro linear_rot 4 ; have_out, need_in, dst, out
+    %if %1 || %2 ; we also need to rotate pure input registers
+        %if %1
+            psllw %4, 8
+        %else
+            psllw %4, %3, 8
+        %endif
+            psrlw %3, 8
+            por %3, %4
+    %endif
+%endmacro
+
+%macro linear_pass 0-1 ; suffix
+%xdefine USED(J) (!(ZERO_MASK & LIN_COL(J)))
+%xdefine IN0 mx%1
+%xdefine IN1 my%1
+%xdefine IN2 mz%1
+%xdefine IN3 mw%1
+
+IF1 X,  linear_row m8,  tmp0q +  0 * 16, 0
+IF1 Y,  linear_row m9,  tmp0q +  5 * 16, 1
+IF1 Z,  linear_row m10, tmp0q + 10 * 16, 2
+IF1 W,  linear_row m11, tmp0q + 15 * 16, 3
+
+    %if BITS == 8
+        ; swap high/low bits and compute the other half; this discards the
+        ; garbage high byte produced by each sub-pass
+        linear_rot X, USED(0), IN0, m8
+        linear_rot Y, USED(1), IN1, m9
+        linear_rot Z, USED(2), IN2, m10
+        linear_rot W, USED(3), IN3, m11
+IF1 X,  linear_row m8,  tmp0q +  0 * 16, 0
+IF1 Y,  linear_row m9,  tmp0q +  5 * 16, 1
+IF1 Z,  linear_row m10, tmp0q + 10 * 16, 2
+IF1 W,  linear_row m11, tmp0q + 15 * 16, 3
+        linear_rot X, USED(0), IN0, m8
+        linear_rot Y, USED(1), IN1, m9
+        linear_rot Z, USED(2), IN2, m10
+        linear_rot W, USED(3), IN3, m11
+    %else
+IF X,   mova IN0, m8
+IF Y,   mova IN1, m9
+IF Z,   mova IN2, m10
+IF W,   mova IN3, m11
+    %endif
+%endmacro
+
+%macro LINEAR 2
+%assign ONE_MASK   %1
+%assign ZERO_MASK  %2
+
+        mov tmp0q, [implq + SwsOpImpl.priv] ; address of matrix
+        LOAD_CONT tmp1q
+        linear_pass
+IF2 V2, linear_pass 2
+        CONTINUE tmp1q
+%endmacro
+
 ;---------------------------------------------------------
 ; Instantiate above macros to generate all uop kernels

@@ -795,6 +902,7 @@ assert 0, SWS_UOP_DITHER is not implemented for integer types
     DECL_%1_RSHIFT          (RSHIFT)
     DECL_%1_LINEAR_FMA      (LINEAR_FMA)
     DECL_%1_DITHER          (DITHER)
+    DECL_%1_LINEAR          (LINEAR)
 %endmacro

 %macro decl_type_invariant 0