Commit f68e1afc1b for ffmpeg
commit f68e1afc1b7cf4275d09f1a9026ff80228cf99a8
Author: Zuxy Meng <zuxy.meng@gmail.com>
Date: Fri Sep 4 21:35:36 2026 -0700
avcodec/loongarch/h264dsp: Fix edge cases for bi-weight on loongarch
S16 saturating sum must be computed dot-product-first. The LSX/LASX code
accumulated the offset into the dot product with wrapping vmaddwev/wod,
so large-magnitude sums wrapped instead of saturating.
Accumulate the dot product from zero (wrapping is exact: the dot product
cannot overflow S16 for 8-bit with log2_denom < 7), then add the offset
with a saturating vsadd.
Also tweaked the checkasm test itself to cover only valid width and height
combo. This fixes bi-weight checkasm test for all currently available
asm implementations.
Signed-off-by: Zuxy Meng <zuxy.meng@gmail.com>
diff --git a/libavcodec/loongarch/h264dsp.S b/libavcodec/loongarch/h264dsp.S
index 750fe49143..e7d4892cf3 100644
--- a/libavcodec/loongarch/h264dsp.S
+++ b/libavcodec/loongarch/h264dsp.S
@@ -943,10 +943,10 @@ endfunc
.macro biweight_calc _in0, _in1, _in2, _in3, _reg0, _reg1, _reg2,\
_out0, _out1, _out2, _out3
- vmov \_out0, \_reg0
- vmov \_out1, \_reg0
- vmov \_out2, \_reg0
- vmov \_out3, \_reg0
+ vreplgr2vr.h \_out0, zero
+ vreplgr2vr.h \_out1, zero
+ vreplgr2vr.h \_out2, zero
+ vreplgr2vr.h \_out3, zero
vmaddwev.h.bu.b \_out0, \_in0, \_reg1
vmaddwev.h.bu.b \_out1, \_in1, \_reg1
vmaddwev.h.bu.b \_out2, \_in2, \_reg1
@@ -955,6 +955,10 @@ endfunc
vmaddwod.h.bu.b \_out1, \_in1, \_reg1
vmaddwod.h.bu.b \_out2, \_in2, \_reg1
vmaddwod.h.bu.b \_out3, \_in3, \_reg1
+ vsadd.h \_out0, \_out0, \_reg0
+ vsadd.h \_out1, \_out1, \_reg0
+ vsadd.h \_out2, \_out2, \_reg0
+ vsadd.h \_out3, \_out3, \_reg0
vssran.bu.h \_out0, \_out0, \_reg2
vssran.bu.h \_out1, \_out1, \_reg2
@@ -1133,9 +1137,10 @@ biweight_func 16
endfunc
.macro biweight_calc_4 _in0, _out0
- vmov \_out0, vr8
+ vreplgr2vr.h \_out0, zero
vmaddwev.h.bu.b \_out0, \_in0, vr20
vmaddwod.h.bu.b \_out0, \_in0, vr20
+ vsadd.h \_out0, \_out0, vr8
vssran.bu.h \_out0, \_out0, vr9
.endm
@@ -1188,12 +1193,14 @@ biweight_func 4
vilvl.b vr0, vr14, vr4
vilvl.b vr10, vr15, vr5
- vmov vr1, vr8
- vmov vr11, vr8
+ vreplgr2vr.h vr1, zero
+ vreplgr2vr.h vr11, zero
vmaddwev.h.bu.b vr1, vr0, vr20
vmaddwev.h.bu.b vr11, vr10, vr20
vmaddwod.h.bu.b vr1, vr0, vr20
vmaddwod.h.bu.b vr11, vr10, vr20
+ vsadd.h vr1, vr1, vr8
+ vsadd.h vr11, vr11, vr8
vssran.bu.h vr0, vr1, vr9 //vec0
vssran.bu.h vr10, vr11, vr9 //vec0
@@ -1225,12 +1232,14 @@ function ff_biweight_h264_pixels\w\()_8_lasx
.endm
.macro biweight_calc_lasx _in0, _in1, _reg0, _reg1, _reg2, _out0, _out1
- xmov \_out0, \_reg0
- xmov \_out1, \_reg0
+ xvreplgr2vr.h \_out0, zero
+ xvreplgr2vr.h \_out1, zero
xvmaddwev.h.bu.b \_out0, \_in0, \_reg1
xvmaddwev.h.bu.b \_out1, \_in1, \_reg1
xvmaddwod.h.bu.b \_out0, \_in0, \_reg1
xvmaddwod.h.bu.b \_out1, \_in1, \_reg1
+ xvsadd.h \_out0, \_out0, \_reg0
+ xvsadd.h \_out1, \_out1, \_reg0
xvssran.bu.h \_out0, \_out0, \_reg2
xvssran.bu.h \_out1, \_out1, \_reg2
diff --git a/tests/checkasm/h264dsp.c b/tests/checkasm/h264dsp.c
index c410461f51..3bd0dec2c3 100644
--- a/tests/checkasm/h264dsp.c
+++ b/tests/checkasm/h264dsp.c
@@ -517,7 +517,8 @@ static void check_weight(void)
if (check_func(h.weight_pixels_tab[idx], "weight_%dx%d_%d",
w, 16, bit_depth)) {
- for (int hgt = 16; hgt >= 2; hgt >>= 1) {
+ for (int hgt = FFMIN(16, 2 * w);
+ hgt >= FFMAX(2, w / 2); hgt >>= 1) {
for (int i = 0; i < 32; i++) {
int stride = 32 * SIZEOF_PIXEL;
int log2_denom = rnd() % 8;
@@ -544,10 +545,6 @@ static void check_weight(void)
}
}
-// only archs that can pass test
-#define H264_CHECK_BIWEIGHT (ARCH_X86 || ARCH_PPC || ARCH_MIPS || ARCH_ARM || ARCH_AARCH64 || ARCH_RISCV)
-
-#if H264_CHECK_BIWEIGHT
static void check_biweight(void)
{
LOCAL_ALIGNED_16(uint8_t, dst, [32 * 32 * 2]);
@@ -568,7 +565,8 @@ static void check_biweight(void)
if (check_func(h.biweight_pixels_tab[idx], "biweight_%dx%d_%d",
w, 16, bit_depth)) {
- for (int hgt = 16; hgt >= 2; hgt >>= 1) {
+ for (int hgt = FFMIN(16, 2 * w);
+ hgt >= FFMAX(2, w / 2); hgt >>= 1) {
for (int i = 0; i < 32; i++) {
int stride = 32 * SIZEOF_PIXEL;
// Spec allows for 0 <= log2_denom <= 7 regardless
@@ -614,7 +612,6 @@ static void check_biweight(void)
}
}
}
-#endif
void checkasm_check_h264dsp(void)
{
@@ -632,8 +629,6 @@ void checkasm_check_h264dsp(void)
check_weight();
report("weight");
-#if H264_CHECK_BIWEIGHT
check_biweight();
report("biweight");
-#endif
}