Commit ce807952aa for ffmpeg
commit ce807952aa13f1aa38eb3d8f6b62ef79ad928c2a
Author: Timo Rothenpieler <timo@rothenpieler.org>
Date: Wed Oct 7 02:26:10 2026 +0200
swscale/rgb2rgb: round the chroma average in uyvy/yuyv to yuv420p
The C and aarch64 implementations of uyvytoyuv420 and yuyvtoyuv420
average the chroma of each line pair with (a + b) >> 1, while the x86
SIMD loop uses pavgb, which computes (a + b + 1) >> 1. Its scalar tail
then truncated again, so x86 output mixed both.
Use round-to-nearest everywhere: in the C reference, the x86 scalar
tail, and on aarch64 by switching uhadd to urhadd and adding the
rounding bias in the scalar paths.
This changes the output of unscaled yuyv422/uyvy422 to yuv420p
conversion. No FATE test covers it.
Co-Authored-By: James Almer <jamrial@gmail.com>
Assisted-by: Claude Opus 5.5
diff --git a/libswscale/aarch64/rgb2rgb_neon.S b/libswscale/aarch64/rgb2rgb_neon.S
index 76a9e07774..2d1707272e 100644
--- a/libswscale/aarch64/rgb2rgb_neon.S
+++ b/libswscale/aarch64/rgb2rgb_neon.S
@@ -644,11 +644,11 @@ w17 - set to 1 if last line has to be handled separately (odd height)
.ifc \dst_fmt, yuv420 // store UV
.ifc \src_fmt, uyvy
- uhadd v0.16b, v4.16b, v0.16b // halving sum of U
- uhadd v2.16b, v6.16b, v2.16b // halving sum of V
+ urhadd v0.16b, v4.16b, v0.16b // halving sum of U
+ urhadd v2.16b, v6.16b, v2.16b // halving sum of V
.else
- uhadd v1.16b, v5.16b, v1.16b // halving sum of U
- uhadd v3.16b, v7.16b, v3.16b // halving sum of V
+ urhadd v1.16b, v5.16b, v1.16b // halving sum of U
+ urhadd v3.16b, v7.16b, v3.16b // halving sum of V
.endif
.endif
@@ -738,6 +738,7 @@ w17 - set to 1 if last line has to be handled separately (odd height)
ldrb w12, [x3], #1
ldrb w14, [x13], #1
add w12, w12, w14
+ add w12, w12, #1
lsr w12, w12, #1
strb w12, [x1], #1
ldrb w14, [x3], #1
@@ -747,6 +748,7 @@ w17 - set to 1 if last line has to be handled separately (odd height)
ldrb w14, [x13], #1
ldrb w12, [x3], #1
add w12, w12, w14
+ add w12, w12, #1
lsr w12, w12, #1
strb w12, [x2], #1
ldrb w14, [x3], #1
@@ -768,6 +770,7 @@ w17 - set to 1 if last line has to be handled separately (odd height)
ldrb w12, [x3], #1
ldrb w14, [x13], #1
add w12, w12, w14
+ add w12, w12, #1
lsr w12, w12, #1
strb w12, [x1], #1
ldrb w14, [x3], #1
@@ -777,6 +780,7 @@ w17 - set to 1 if last line has to be handled separately (odd height)
ldrb w14, [x13], #1
ldrb w12, [x3], #1
add w12, w12, w14
+ add w12, w12, #1
lsr w12, w12, #1
strb w12, [x2], #1
.endif
@@ -829,14 +833,16 @@ w17 - set to 1 if last line has to be handled separately (odd height)
strb w12, [x0]
ldrb w12, [x13, #1] // Y, bottom line
strb w12, [x10]
- ldrb w12, [x3] // U = (top + bottom) >> 1
+ ldrb w12, [x3] // U = (top + bottom + 1) >> 1
ldrb w14, [x13]
add w12, w12, w14
+ add w12, w12, #1
lsr w12, w12, #1
strb w12, [x1]
- ldrb w12, [x3, #2] // V = (top + bottom) >> 1
+ ldrb w12, [x3, #2] // V = (top + bottom + 1) >> 1
ldrb w14, [x13, #2]
add w12, w12, w14
+ add w12, w12, #1
lsr w12, w12, #1
strb w12, [x2]
.else
@@ -844,14 +850,16 @@ w17 - set to 1 if last line has to be handled separately (odd height)
strb w12, [x0]
ldrb w12, [x13] // Y, bottom line
strb w12, [x10]
- ldrb w12, [x3, #1] // U = (top + bottom) >> 1
+ ldrb w12, [x3, #1] // U = (top + bottom + 1) >> 1
ldrb w14, [x13, #1]
add w12, w12, w14
+ add w12, w12, #1
lsr w12, w12, #1
strb w12, [x1]
- ldrb w12, [x3, #3] // V = (top + bottom) >> 1
+ ldrb w12, [x3, #3] // V = (top + bottom + 1) >> 1
ldrb w14, [x13, #3]
add w12, w12, w14
+ add w12, w12, #1
lsr w12, w12, #1
strb w12, [x2]
.endif
diff --git a/libswscale/rgb2rgb_template.c b/libswscale/rgb2rgb_template.c
index 1f0aef1fb9..df439783d6 100644
--- a/libswscale/rgb2rgb_template.c
+++ b/libswscale/rgb2rgb_template.c
@@ -710,8 +710,8 @@ static void extract_even2avg_c(const uint8_t *src0, const uint8_t *src1,
src1 += count * 4;
count = -count;
while (count < 0) {
- dst0[count] = (src0[4 * count + 0] + src1[4 * count + 0]) >> 1;
- dst1[count] = (src0[4 * count + 2] + src1[4 * count + 2]) >> 1;
+ dst0[count] = (src0[4 * count + 0] + src1[4 * count + 0] + 1) >> 1;
+ dst1[count] = (src0[4 * count + 2] + src1[4 * count + 2] + 1) >> 1;
count++;
}
}
@@ -742,8 +742,8 @@ static void extract_odd2avg_c(const uint8_t *src0, const uint8_t *src1,
src0++;
src1++;
while (count < 0) {
- dst0[count] = (src0[4 * count + 0] + src1[4 * count + 0]) >> 1;
- dst1[count] = (src0[4 * count + 2] + src1[4 * count + 2]) >> 1;
+ dst0[count] = (src0[4 * count + 0] + src1[4 * count + 0] + 1) >> 1;
+ dst1[count] = (src0[4 * count + 2] + src1[4 * count + 2] + 1) >> 1;
count++;
}
}
diff --git a/libswscale/x86/rgb2rgb.c b/libswscale/x86/rgb2rgb.c
index 16994f2023..699d341ce9 100644
--- a/libswscale/x86/rgb2rgb.c
+++ b/libswscale/x86/rgb2rgb.c
@@ -1746,8 +1746,8 @@ static void extract_even2avg_mmxext(const uint8_t *src0, const uint8_t *src1, ui
}
#endif
while(count<0) {
- dst0[count]= (src0[4*count+0]+src1[4*count+0])>>1;
- dst1[count]= (src0[4*count+2]+src1[4*count+2])>>1;
+ dst0[count] = (src0[4*count+0]+src1[4*count+0]+1)>>1;
+ dst1[count] = (src0[4*count+2]+src1[4*count+2]+1)>>1;
count++;
}
}
@@ -1848,8 +1848,8 @@ static void extract_odd2avg_mmxext(const uint8_t *src0, const uint8_t *src1, uin
src0++;
src1++;
while(count<0) {
- dst0[count]= (src0[4*count+0]+src1[4*count+0])>>1;
- dst1[count]= (src0[4*count+2]+src1[4*count+2])>>1;
+ dst0[count] = (src0[4*count+0]+src1[4*count+0]+1)>>1;
+ dst1[count] = (src0[4*count+2]+src1[4*count+2]+1)>>1;
count++;
}
}