Commit f68e1afc1b for ffmpeg

commit f68e1afc1b7cf4275d09f1a9026ff80228cf99a8
Author: Zuxy Meng <zuxy.meng@gmail.com>
Date:   Fri Sep 4 21:35:36 2026 -0700

    avcodec/loongarch/h264dsp: Fix edge cases for bi-weight on loongarch

    S16 saturating sum must be computed dot-product-first. The LSX/LASX code
    accumulated the offset into the dot product with wrapping vmaddwev/wod,
    so large-magnitude sums wrapped instead of saturating.

    Accumulate the dot product from zero (wrapping is exact: the dot product
    cannot overflow S16 for 8-bit with log2_denom < 7), then add the offset
    with a saturating vsadd.

    Also tweaked the checkasm test itself to cover only valid width and height
    combo.  This fixes bi-weight checkasm test for all currently available
    asm implementations.

    Signed-off-by: Zuxy Meng <zuxy.meng@gmail.com>

diff --git a/libavcodec/loongarch/h264dsp.S b/libavcodec/loongarch/h264dsp.S
index 750fe49143..e7d4892cf3 100644
--- a/libavcodec/loongarch/h264dsp.S
+++ b/libavcodec/loongarch/h264dsp.S
@@ -943,10 +943,10 @@ endfunc

 .macro biweight_calc _in0, _in1, _in2, _in3, _reg0, _reg1, _reg2,\
                      _out0, _out1, _out2, _out3
-    vmov             \_out0,   \_reg0
-    vmov             \_out1,   \_reg0
-    vmov             \_out2,   \_reg0
-    vmov             \_out3,   \_reg0
+    vreplgr2vr.h     \_out0,   zero
+    vreplgr2vr.h     \_out1,   zero
+    vreplgr2vr.h     \_out2,   zero
+    vreplgr2vr.h     \_out3,   zero
     vmaddwev.h.bu.b  \_out0,   \_in0,     \_reg1
     vmaddwev.h.bu.b  \_out1,   \_in1,     \_reg1
     vmaddwev.h.bu.b  \_out2,   \_in2,     \_reg1
@@ -955,6 +955,10 @@ endfunc
     vmaddwod.h.bu.b  \_out1,   \_in1,     \_reg1
     vmaddwod.h.bu.b  \_out2,   \_in2,     \_reg1
     vmaddwod.h.bu.b  \_out3,   \_in3,     \_reg1
+    vsadd.h          \_out0,   \_out0,    \_reg0
+    vsadd.h          \_out1,   \_out1,    \_reg0
+    vsadd.h          \_out2,   \_out2,    \_reg0
+    vsadd.h          \_out3,   \_out3,    \_reg0

     vssran.bu.h      \_out0,   \_out0,    \_reg2
     vssran.bu.h      \_out1,   \_out1,    \_reg2
@@ -1133,9 +1137,10 @@ biweight_func 16
 endfunc

 .macro biweight_calc_4 _in0, _out0
-    vmov             \_out0,  vr8
+    vreplgr2vr.h     \_out0,  zero
     vmaddwev.h.bu.b  \_out0,  \_in0,  vr20
     vmaddwod.h.bu.b  \_out0,  \_in0,  vr20
+    vsadd.h          \_out0,  \_out0, vr8
     vssran.bu.h      \_out0,  \_out0, vr9
 .endm

@@ -1188,12 +1193,14 @@ biweight_func 4
     vilvl.b          vr0,    vr14,   vr4
     vilvl.b          vr10,   vr15,   vr5

-    vmov             vr1,    vr8
-    vmov             vr11,   vr8
+    vreplgr2vr.h     vr1,    zero
+    vreplgr2vr.h     vr11,   zero
     vmaddwev.h.bu.b  vr1,    vr0,    vr20
     vmaddwev.h.bu.b  vr11,   vr10,   vr20
     vmaddwod.h.bu.b  vr1,    vr0,    vr20
     vmaddwod.h.bu.b  vr11,   vr10,   vr20
+    vsadd.h          vr1,    vr1,    vr8
+    vsadd.h          vr11,   vr11,   vr8

     vssran.bu.h      vr0,    vr1,    vr9   //vec0
     vssran.bu.h      vr10,   vr11,   vr9   //vec0
@@ -1225,12 +1232,14 @@ function ff_biweight_h264_pixels\w\()_8_lasx
 .endm

 .macro biweight_calc_lasx _in0, _in1, _reg0, _reg1, _reg2, _out0, _out1
-    xmov              \_out0,   \_reg0
-    xmov              \_out1,   \_reg0
+    xvreplgr2vr.h     \_out0,   zero
+    xvreplgr2vr.h     \_out1,   zero
     xvmaddwev.h.bu.b  \_out0,   \_in0,     \_reg1
     xvmaddwev.h.bu.b  \_out1,   \_in1,     \_reg1
     xvmaddwod.h.bu.b  \_out0,   \_in0,     \_reg1
     xvmaddwod.h.bu.b  \_out1,   \_in1,     \_reg1
+    xvsadd.h          \_out0,   \_out0,    \_reg0
+    xvsadd.h          \_out1,   \_out1,    \_reg0

     xvssran.bu.h      \_out0,   \_out0,    \_reg2
     xvssran.bu.h      \_out1,   \_out1,    \_reg2
diff --git a/tests/checkasm/h264dsp.c b/tests/checkasm/h264dsp.c
index c410461f51..3bd0dec2c3 100644
--- a/tests/checkasm/h264dsp.c
+++ b/tests/checkasm/h264dsp.c
@@ -517,7 +517,8 @@ static void check_weight(void)

             if (check_func(h.weight_pixels_tab[idx], "weight_%dx%d_%d",
                            w, 16, bit_depth)) {
-                for (int hgt = 16; hgt >= 2; hgt >>= 1) {
+                for (int hgt = FFMIN(16, 2 * w);
+                     hgt >= FFMAX(2, w / 2); hgt >>= 1) {
                     for (int i = 0; i < 32; i++) {
                         int stride = 32 * SIZEOF_PIXEL;
                         int log2_denom = rnd() % 8;
@@ -544,10 +545,6 @@ static void check_weight(void)
     }
 }

-// only archs that can pass test
-#define H264_CHECK_BIWEIGHT (ARCH_X86 || ARCH_PPC || ARCH_MIPS || ARCH_ARM || ARCH_AARCH64 || ARCH_RISCV)
-
-#if H264_CHECK_BIWEIGHT
 static void check_biweight(void)
 {
     LOCAL_ALIGNED_16(uint8_t, dst, [32 * 32 * 2]);
@@ -568,7 +565,8 @@ static void check_biweight(void)

             if (check_func(h.biweight_pixels_tab[idx], "biweight_%dx%d_%d",
                            w, 16, bit_depth)) {
-                for (int hgt = 16; hgt >= 2; hgt >>= 1) {
+                for (int hgt = FFMIN(16, 2 * w);
+                     hgt >= FFMAX(2, w / 2); hgt >>= 1) {
                     for (int i = 0; i < 32; i++) {
                         int stride = 32 * SIZEOF_PIXEL;
                         // Spec allows for 0 <= log2_denom <= 7 regardless
@@ -614,7 +612,6 @@ static void check_biweight(void)
         }
     }
 }
-#endif

 void checkasm_check_h264dsp(void)
 {
@@ -632,8 +629,6 @@ void checkasm_check_h264dsp(void)
     check_weight();
     report("weight");

-#if H264_CHECK_BIWEIGHT
     check_biweight();
     report("biweight");
-#endif
 }