Commit ce807952aa for ffmpeg

commit ce807952aa13f1aa38eb3d8f6b62ef79ad928c2a
Author: Timo Rothenpieler <timo@rothenpieler.org>
Date:   Wed Oct 7 02:26:10 2026 +0200

    swscale/rgb2rgb: round the chroma average in uyvy/yuyv to yuv420p

    The C and aarch64 implementations of uyvytoyuv420 and yuyvtoyuv420
    average the chroma of each line pair with (a + b) >> 1, while the x86
    SIMD loop uses pavgb, which computes (a + b + 1) >> 1. Its scalar tail
    then truncated again, so x86 output mixed both.

    Use round-to-nearest everywhere: in the C reference, the x86 scalar
    tail, and on aarch64 by switching uhadd to urhadd and adding the
    rounding bias in the scalar paths.

    This changes the output of unscaled yuyv422/uyvy422 to yuv420p
    conversion. No FATE test covers it.

    Co-Authored-By: James Almer <jamrial@gmail.com>
    Assisted-by: Claude Opus 5.5

diff --git a/libswscale/aarch64/rgb2rgb_neon.S b/libswscale/aarch64/rgb2rgb_neon.S
index 76a9e07774..2d1707272e 100644
--- a/libswscale/aarch64/rgb2rgb_neon.S
+++ b/libswscale/aarch64/rgb2rgb_neon.S
@@ -644,11 +644,11 @@ w17 - set to 1 if last line has to be handled separately (odd height)

 .ifc \dst_fmt, yuv420                                    // store UV
 .ifc \src_fmt, uyvy
-        uhadd           v0.16b, v4.16b, v0.16b            // halving sum of U
-        uhadd           v2.16b, v6.16b, v2.16b            // halving sum of V
+        urhadd          v0.16b, v4.16b, v0.16b            // halving sum of U
+        urhadd          v2.16b, v6.16b, v2.16b            // halving sum of V
 .else
-        uhadd           v1.16b, v5.16b, v1.16b            // halving sum of U
-        uhadd           v3.16b, v7.16b, v3.16b            // halving sum of V
+        urhadd          v1.16b, v5.16b, v1.16b            // halving sum of U
+        urhadd          v3.16b, v7.16b, v3.16b            // halving sum of V
 .endif
 .endif

@@ -738,6 +738,7 @@ w17 - set to 1 if last line has to be handled separately (odd height)
         ldrb            w12, [x3], #1
         ldrb            w14, [x13], #1
         add             w12, w12, w14
+        add             w12, w12, #1
         lsr             w12, w12, #1
         strb            w12, [x1], #1
         ldrb            w14, [x3], #1
@@ -747,6 +748,7 @@ w17 - set to 1 if last line has to be handled separately (odd height)
         ldrb            w14, [x13], #1
         ldrb            w12, [x3], #1
         add             w12, w12, w14
+        add             w12, w12, #1
         lsr             w12, w12, #1
         strb            w12, [x2], #1
         ldrb            w14, [x3], #1
@@ -768,6 +770,7 @@ w17 - set to 1 if last line has to be handled separately (odd height)
         ldrb            w12, [x3], #1
         ldrb            w14, [x13], #1
         add             w12, w12, w14
+        add             w12, w12, #1
         lsr             w12, w12, #1
         strb            w12, [x1], #1
         ldrb            w14, [x3], #1
@@ -777,6 +780,7 @@ w17 - set to 1 if last line has to be handled separately (odd height)
         ldrb            w14, [x13], #1
         ldrb            w12, [x3], #1
         add             w12, w12, w14
+        add             w12, w12, #1
         lsr             w12, w12, #1
         strb            w12, [x2], #1
 .endif
@@ -829,14 +833,16 @@ w17 - set to 1 if last line has to be handled separately (odd height)
         strb            w12, [x0]
         ldrb            w12, [x13, #1]                    // Y, bottom line
         strb            w12, [x10]
-        ldrb            w12, [x3]                         // U = (top + bottom) >> 1
+        ldrb            w12, [x3]                         // U = (top + bottom + 1) >> 1
         ldrb            w14, [x13]
         add             w12, w12, w14
+        add             w12, w12, #1
         lsr             w12, w12, #1
         strb            w12, [x1]
-        ldrb            w12, [x3, #2]                     // V = (top + bottom) >> 1
+        ldrb            w12, [x3, #2]                     // V = (top + bottom + 1) >> 1
         ldrb            w14, [x13, #2]
         add             w12, w12, w14
+        add             w12, w12, #1
         lsr             w12, w12, #1
         strb            w12, [x2]
 .else
@@ -844,14 +850,16 @@ w17 - set to 1 if last line has to be handled separately (odd height)
         strb            w12, [x0]
         ldrb            w12, [x13]                        // Y, bottom line
         strb            w12, [x10]
-        ldrb            w12, [x3, #1]                     // U = (top + bottom) >> 1
+        ldrb            w12, [x3, #1]                     // U = (top + bottom + 1) >> 1
         ldrb            w14, [x13, #1]
         add             w12, w12, w14
+        add             w12, w12, #1
         lsr             w12, w12, #1
         strb            w12, [x1]
-        ldrb            w12, [x3, #3]                     // V = (top + bottom) >> 1
+        ldrb            w12, [x3, #3]                     // V = (top + bottom + 1) >> 1
         ldrb            w14, [x13, #3]
         add             w12, w12, w14
+        add             w12, w12, #1
         lsr             w12, w12, #1
         strb            w12, [x2]
 .endif
diff --git a/libswscale/rgb2rgb_template.c b/libswscale/rgb2rgb_template.c
index 1f0aef1fb9..df439783d6 100644
--- a/libswscale/rgb2rgb_template.c
+++ b/libswscale/rgb2rgb_template.c
@@ -710,8 +710,8 @@ static void extract_even2avg_c(const uint8_t *src0, const uint8_t *src1,
     src1  +=  count * 4;
     count  = -count;
     while (count < 0) {
-        dst0[count] = (src0[4 * count + 0] + src1[4 * count + 0]) >> 1;
-        dst1[count] = (src0[4 * count + 2] + src1[4 * count + 2]) >> 1;
+        dst0[count] = (src0[4 * count + 0] + src1[4 * count + 0] + 1) >> 1;
+        dst1[count] = (src0[4 * count + 2] + src1[4 * count + 2] + 1) >> 1;
         count++;
     }
 }
@@ -742,8 +742,8 @@ static void extract_odd2avg_c(const uint8_t *src0, const uint8_t *src1,
     src0++;
     src1++;
     while (count < 0) {
-        dst0[count] = (src0[4 * count + 0] + src1[4 * count + 0]) >> 1;
-        dst1[count] = (src0[4 * count + 2] + src1[4 * count + 2]) >> 1;
+        dst0[count] = (src0[4 * count + 0] + src1[4 * count + 0] + 1) >> 1;
+        dst1[count] = (src0[4 * count + 2] + src1[4 * count + 2] + 1) >> 1;
         count++;
     }
 }
diff --git a/libswscale/x86/rgb2rgb.c b/libswscale/x86/rgb2rgb.c
index 16994f2023..699d341ce9 100644
--- a/libswscale/x86/rgb2rgb.c
+++ b/libswscale/x86/rgb2rgb.c
@@ -1746,8 +1746,8 @@ static void extract_even2avg_mmxext(const uint8_t *src0, const uint8_t *src1, ui
     }
 #endif
     while(count<0) {
-        dst0[count]= (src0[4*count+0]+src1[4*count+0])>>1;
-        dst1[count]= (src0[4*count+2]+src1[4*count+2])>>1;
+        dst0[count] = (src0[4*count+0]+src1[4*count+0]+1)>>1;
+        dst1[count] = (src0[4*count+2]+src1[4*count+2]+1)>>1;
         count++;
     }
 }
@@ -1848,8 +1848,8 @@ static void extract_odd2avg_mmxext(const uint8_t *src0, const uint8_t *src1, uin
     src0++;
     src1++;
     while(count<0) {
-        dst0[count]= (src0[4*count+0]+src1[4*count+0])>>1;
-        dst1[count]= (src0[4*count+2]+src1[4*count+2])>>1;
+        dst0[count] = (src0[4*count+0]+src1[4*count+0]+1)>>1;
+        dst1[count] = (src0[4*count+2]+src1[4*count+2]+1)>>1;
         count++;
     }
 }