Commit b91a82d6dd for ffmpeg

commit b91a82d6dd0e7c32caf34b345130c8fb1309a597
Author: jinbo <jinbo@loongson.cn>
Date:   Tue Sep 1 11:28:33 2026 +0800

    swscale/loongarch: add LSX fast bilinear horizontal scale

    Add LoongArch LSX optimized implementations of the fast bilinear
    horizontal scaler (SWS_FAST_BILINEAR, hyscale_fast/hcscale_fast).

    The vectorized loop processes 8 destination pixels per iteration,
    computing the source offsets and interpolation weights in registers
    and gathering the input samples with vshuf_b over a 32-byte window.
    It covers scaling ratios down to 4x downscaling; beyond that the
    scalar loop is used.

    checkasm --bench on 3A5000 4 cores 2.5GHz:

    hcscale_fast_c:          99.7
    hcscale_fast_lsx:        28.1 ( 3.55x)
    hyscale_fast_c:          51.1
    hyscale_fast_lsx:        18.3 ( 2.78x)

    Performance with:
    $ ./ffmpeg -cpuflags lsx -f lavfi -i "smptebars=size=3840x2160:rate=30:duration=100,format=nv12" \
    -threads 1 -vf "scale=1920:1080:sws_flags=fast_bilinear,format=bgra" -f null -
    before: 36fps
    after : 86fps

diff --git a/libswscale/loongarch/Makefile b/libswscale/loongarch/Makefile
index 06aed9d245..0b9c87100e 100644
--- a/libswscale/loongarch/Makefile
+++ b/libswscale/loongarch/Makefile
@@ -6,6 +6,7 @@ LASX-OBJS-$(CONFIG_SWSCALE) += loongarch/swscale_lasx.o \
                                loongarch/output_lasx.o
 LSX-OBJS-$(CONFIG_SWSCALE)  += loongarch/swscale.o \
                                loongarch/swscale_unscaled.o \
+                               loongarch/hscale_fast_bilinear_lsx.o \
                                loongarch/swscale_lsx.o \
                                loongarch/input.o   \
                                loongarch/output.o  \
diff --git a/libswscale/loongarch/hscale_fast_bilinear_lsx.c b/libswscale/loongarch/hscale_fast_bilinear_lsx.c
new file mode 100644
index 0000000000..7896542251
--- /dev/null
+++ b/libswscale/loongarch/hscale_fast_bilinear_lsx.c
@@ -0,0 +1,162 @@
+/*
+ * Copyright (C) 2026 Loongson Technology Co. Ltd.
+ * Contributed by Bo Jin(jinbo@loongson.cn)
+ *
+ * This file is part of FFmpeg.
+ *
+ * FFmpeg is free software; you can redistribute it and/or
+ * modify it under the terms of the GNU Lesser General Public
+ * License as published by the Free Software Foundation; either
+ * version 2.1 of the License, or (at your option) any later version.
+ *
+ * FFmpeg is distributed in the hope that it will be useful,
+ * but WITHOUT ANY WARRANTY; without even the implied warranty of
+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
+ * Lesser General Public License for more details.
+ *
+ * You should have received a copy of the GNU Lesser General Public
+ * License along with FFmpeg; if not, write to the Free Software
+ * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
+ */
+
+#include "swscale_loongarch.h"
+#include "libavutil/loongarch/loongson_intrinsics.h"
+
+/* Each vector iteration scales 8 destination pixels. Their source
+ * position offsets ((xpos & 0xFFFF) + j*xInc) >> 16, j in [0, 7], must
+ * stay within the 32-byte gather window, which holds while
+ * xInc <= (1 << 18); otherwise fall back to the scalar loop. */
+#define LSX_HSCALE_FAST_MAX_XINC (1 << 18)
+
+void ff_hyscale_fast_lsx(SwsInternal *c, int16_t *dst, int dstWidth,
+                         const uint8_t *src, int srcW, int xInc)
+{
+    int i = 0;
+    unsigned int xpos = 0;
+
+    if (xInc <= LSX_HSCALE_FAST_MAX_XINC) {
+        static const int32_t idx32[4]  = {0, 1, 2, 3};
+        static const int16_t idx16[8]  = {0, 1, 2, 3, 4, 5, 6, 7};
+        static const uint8_t shuf8[16] = {
+            0, 4, 8, 12, 16, 20, 24, 28,
+            0x30, 0x30, 0x30, 0x30, 0x30, 0x30, 0x30, 0x30
+        };
+
+        /* [0, xInc, 2*xInc, 3*xInc] as 32-bit lanes */
+        __m128i vadd_w = __lsx_vmul_w(__lsx_vreplgr2vr_w(xInc),
+                                      __lsx_vld(idx32, 0));
+        /* [0, xInc, ..., 7*xInc] modulo 2^16 as 16-bit lanes */
+        __m128i vadd16 = __lsx_vmul_h(__lsx_vreplgr2vr_h(xInc),
+                                      __lsx_vld(idx16, 0));
+        __m128i vx4    = __lsx_vreplgr2vr_w(4 * xInc);
+        __m128i vshuf8 = __lsx_vld(shuf8, 0);
+        __m128i v128   = __lsx_vreplgr2vr_h(128);
+
+        for (; i + 8 <= dstWidth && (xpos >> 16) + 32 < srcW; i += 8, xpos += 8 * xInc) {
+            unsigned int lo = xpos & 0xFFFF;
+            unsigned int xx = xpos >> 16;
+
+            /* full 32-bit positions of the 8 pixels */
+            __m128i vc0 = __lsx_vadd_w(__lsx_vreplgr2vr_w(lo), vadd_w);
+            __m128i vc1 = __lsx_vadd_w(vc0, vx4);
+
+            /* source offsets (j = 0..7), each in [0, 28] */
+            __m128i vperm = __lsx_vshuf_b(__lsx_vsrli_w(vc1, 16),
+                                          __lsx_vsrli_w(vc0, 16), vshuf8);
+
+            /* xalpha = (xpos & 0xFFFF) >> 9, 16-bit lanes */
+            __m128i valpha = __lsx_vsrli_h(__lsx_vadd_h(__lsx_vreplgr2vr_h(lo),
+                                                        vadd16), 9);
+
+            __m128i v0 = __lsx_vsllwil_hu_bu(__lsx_vshuf_b(__lsx_vld(src + xx + 16, 0),
+                                                           __lsx_vld(src + xx, 0),
+                                                           vperm), 0);
+            __m128i v1 = __lsx_vsllwil_hu_bu(__lsx_vshuf_b(__lsx_vld(src + xx + 17, 0),
+                                                           __lsx_vld(src + xx + 1, 0),
+                                                           vperm), 0);
+
+            __m128i w0 = __lsx_vsub_h(v128, valpha);
+            __lsx_vst(__lsx_vadd_h(__lsx_vmul_h(v0, w0), __lsx_vmul_h(v1, valpha)),
+                      dst + i, 0);
+        }
+    }
+
+    for (; i < dstWidth; i++) {
+        unsigned int xx     = xpos >> 16;
+        unsigned int xalpha = (xpos & 0xFFFF) >> 9;
+        dst[i] = (src[xx] << 7) + (src[xx + 1] - src[xx]) * xalpha;
+        xpos  += xInc;
+    }
+    for (i = dstWidth - 1; (i * (int64_t)xInc) >> 16 >= srcW - 1; i--)
+        dst[i] = src[srcW - 1] * 128;
+}
+
+void ff_hcscale_fast_lsx(SwsInternal *c, int16_t *dst1, int16_t *dst2,
+                         int dstWidth, const uint8_t *src1,
+                         const uint8_t *src2, int srcW, int xInc)
+{
+    int i = 0;
+    unsigned int xpos = 0;
+
+    if (xInc <= LSX_HSCALE_FAST_MAX_XINC) {
+        static const int32_t idx32[4]  = {0, 1, 2, 3};
+        static const int16_t idx16[8]  = {0, 1, 2, 3, 4, 5, 6, 7};
+        static const uint8_t shuf8[16] = {
+            0, 4, 8, 12, 16, 20, 24, 28,
+            0x30, 0x30, 0x30, 0x30, 0x30, 0x30, 0x30, 0x30
+        };
+
+        __m128i vadd_w = __lsx_vmul_w(__lsx_vreplgr2vr_w(xInc),
+                                      __lsx_vld(idx32, 0));
+        __m128i vadd16 = __lsx_vmul_h(__lsx_vreplgr2vr_h(xInc),
+                                      __lsx_vld(idx16, 0));
+        __m128i vx4    = __lsx_vreplgr2vr_w(4 * xInc);
+        __m128i vshuf8 = __lsx_vld(shuf8, 0);
+        __m128i v127   = __lsx_vreplgr2vr_h(127);
+
+        for (; i + 8 <= dstWidth && (xpos >> 16) + 32 < srcW; i += 8, xpos += 8 * xInc) {
+            unsigned int lo = xpos & 0xFFFF;
+            unsigned int xx = xpos >> 16;
+
+            __m128i vc0 = __lsx_vadd_w(__lsx_vreplgr2vr_w(lo), vadd_w);
+            __m128i vc1 = __lsx_vadd_w(vc0, vx4);
+
+            __m128i vperm = __lsx_vshuf_b(__lsx_vsrli_w(vc1, 16),
+                                          __lsx_vsrli_w(vc0, 16), vshuf8);
+
+            __m128i valpha = __lsx_vsrli_h(__lsx_vadd_h(__lsx_vreplgr2vr_h(lo),
+                                                        vadd16), 9);
+
+            __m128i v10 = __lsx_vsllwil_hu_bu(__lsx_vshuf_b(__lsx_vld(src1 + xx + 16, 0),
+                                                            __lsx_vld(src1 + xx, 0),
+                                                            vperm), 0);
+            __m128i v11 = __lsx_vsllwil_hu_bu(__lsx_vshuf_b(__lsx_vld(src1 + xx + 17, 0),
+                                                            __lsx_vld(src1 + xx + 1, 0),
+                                                            vperm), 0);
+            __m128i v20 = __lsx_vsllwil_hu_bu(__lsx_vshuf_b(__lsx_vld(src2 + xx + 16, 0),
+                                                            __lsx_vld(src2 + xx, 0),
+                                                            vperm), 0);
+            __m128i v21 = __lsx_vsllwil_hu_bu(__lsx_vshuf_b(__lsx_vld(src2 + xx + 17, 0),
+                                                            __lsx_vld(src2 + xx + 1, 0),
+                                                            vperm), 0);
+
+            __m128i w0 = __lsx_vsub_h(v127, valpha);
+            __lsx_vst(__lsx_vadd_h(__lsx_vmul_h(v10, w0), __lsx_vmul_h(v11, valpha)),
+                      dst1 + i, 0);
+            __lsx_vst(__lsx_vadd_h(__lsx_vmul_h(v20, w0), __lsx_vmul_h(v21, valpha)),
+                      dst2 + i, 0);
+        }
+    }
+
+    for (; i < dstWidth; i++) {
+        unsigned int xx     = xpos >> 16;
+        unsigned int xalpha = (xpos & 0xFFFF) >> 9;
+        dst1[i] = (src1[xx] * (xalpha ^ 127) + src1[xx + 1] * xalpha);
+        dst2[i] = (src2[xx] * (xalpha ^ 127) + src2[xx + 1] * xalpha);
+        xpos   += xInc;
+    }
+    for (i = dstWidth - 1; (i * (int64_t)xInc) >> 16 >= srcW - 1; i--) {
+        dst1[i] = src1[srcW - 1] * 128;
+        dst2[i] = src2[srcW - 1] * 128;
+    }
+}
diff --git a/libswscale/loongarch/swscale_init_loongarch.c b/libswscale/loongarch/swscale_init_loongarch.c
index 0c937b047f..ced73f2b0f 100644
--- a/libswscale/loongarch/swscale_init_loongarch.c
+++ b/libswscale/loongarch/swscale_init_loongarch.c
@@ -71,6 +71,12 @@ av_cold void ff_sws_init_swscale_loongarch(SwsInternal *c)
             c->hyScale = c->hcScale = c->dstBpc > 14 ? ff_hscale_16_to_19_lsx
                                                      : ff_hscale_16_to_15_lsx;
         }
+        if (c->srcBpc == 8 && c->dstBpc <= 14 &&
+            c->opts.flags & SWS_FAST_BILINEAR &&
+            c->lumXInc <= (1 << 18) && c->chrXInc <= (1 << 18)) {
+            c->hyscale_fast = ff_hyscale_fast_lsx;
+            c->hcscale_fast = ff_hcscale_fast_lsx;
+        }
     }
 #if HAVE_LASX
     if (have_lasx(cpu_flags)) {
diff --git a/libswscale/loongarch/swscale_loongarch.h b/libswscale/loongarch/swscale_loongarch.h
index a8297b6972..b2875ba70f 100644
--- a/libswscale/loongarch/swscale_loongarch.h
+++ b/libswscale/loongarch/swscale_loongarch.h
@@ -50,6 +50,13 @@ void ff_hscale_16_to_19_sub_lsx(SwsInternal *c, int16_t *_dst, int dstW,
                                 const uint8_t *_src, const int16_t *filter,
                                 const int32_t *filterPos, int filterSize, int sh);

+void ff_hyscale_fast_lsx(SwsInternal *c, int16_t *dst, int dstWidth,
+                         const uint8_t *src, int srcW, int xInc);
+
+void ff_hcscale_fast_lsx(SwsInternal *c, int16_t *dst1, int16_t *dst2,
+                         int dstWidth, const uint8_t *src1,
+                         const uint8_t *src2, int srcW, int xInc);
+
 void lumRangeFromJpeg_lsx(int16_t *dst, int width, uint32_t coeff, int64_t offset);
 void chrRangeFromJpeg_lsx(int16_t *dstU, int16_t *dstV, int width, uint32_t coeff, int64_t offset);
 void lumRangeToJpeg_lsx(int16_t *dst, int width, uint32_t coeff, int64_t offset);
diff --git a/tests/checkasm/sw_scale.c b/tests/checkasm/sw_scale.c
index b06ac23392..68d8f8d735 100644
--- a/tests/checkasm/sw_scale.c
+++ b/tests/checkasm/sw_scale.c
@@ -492,10 +492,130 @@ static void check_hscale(void)
     sws_freeContext(sws);
 }

+/* Fast-bilinear horizontal scaling (SWS_FAST_BILINEAR, c->hyscale_fast /
+ * c->hcscale_fast): the optimized implementation must be bit-exact with
+ * the C reference. Catches e.g. swapped gather-position/weight tables in
+ * the LoongArch LSX implementation, which no FATE frame test covers.
+ *
+ * Only enabled on LoongArch: other archs' fast-bilinear implementations
+ * are not guaranteed bit-exact with the C reference (e.g. x86 MMXEXT uses
+ * complementary weights), so a cross-arch comparison would fail there. */
+#if ARCH_LOONGARCH64
+static void check_hyscale_fast(void)
+{
+#define HSCALE_FAST_SRC_SIZE 4096
+    static const int dstW_list[] = { 8, 16, 63, 100, 255, 256, 300, 511, 512, 1024 };
+    static const int xInc_list[] = { 32768, 65536, 83886, 98304, 131072, 200000, 262143, 262144 };
+    LOCAL_ALIGNED_32(uint8_t, src, [HSCALE_FAST_SRC_SIZE + 32]);
+    LOCAL_ALIGNED_32(int16_t, dst0, [2048]);
+    LOCAL_ALIGNED_32(int16_t, dst1, [2048]);
+    SwsContext *sws;
+    SwsInternal *c;
+    int i, j;
+
+    declare_func(void, SwsInternal *c, int16_t *dst, int dstWidth,
+                 const uint8_t *src, int srcW, int xInc);
+
+    sws = sws_alloc_context();
+    if (!sws || sws_init_context(sws, NULL, NULL) < 0)
+        fail();
+
+    c = sws_internal(sws);
+    c->srcBpc = 8;
+    c->dstBpc = 8;
+    c->opts.flags  = SWS_FAST_BILINEAR;
+    c->lumXInc = c->chrXInc = 83886;
+    ff_sws_init_scale(c);
+
+    if (check_func(c->hyscale_fast, "hyscale_fast")) {
+        for (i = 0; i < FF_ARRAY_ELEMS(dstW_list); i++) {
+            for (j = 0; j < FF_ARRAY_ELEMS(xInc_list); j++) {
+                int dstW = dstW_list[i];
+                int xInc = xInc_list[j];
+                int srcW = FFMIN(HSCALE_FAST_SRC_SIZE,
+                                 (dstW * xInc) >> 16);
+
+                randomize_buffers(src, srcW + 16);
+                memset(dst0, 0, dstW * sizeof(dst0[0]));
+                memset(dst1, 0, dstW * sizeof(dst1[0]));
+
+                call_ref(NULL, dst0, dstW, src, srcW, xInc);
+                call_new(NULL, dst1, dstW, src, srcW, xInc);
+                if (memcmp(dst0, dst1, dstW * sizeof(dst0[0])))
+                    fail();
+            }
+        }
+        bench_new(NULL, dst1, 300, src, 512, 83886);
+    }
+    sws_freeContext(sws);
+}
+
+static void check_hcscale_fast(void)
+{
+#define HCSCALE_FAST_SRC_SIZE 4096
+    static const int dstW_list[] = { 8, 16, 63, 100, 255, 256, 300, 511, 512, 1024 };
+    static const int xInc_list[] = { 32768, 65536, 83886, 98304, 131072, 200000, 262143, 262144 };
+    LOCAL_ALIGNED_32(uint8_t, src1, [HCSCALE_FAST_SRC_SIZE + 32]);
+    LOCAL_ALIGNED_32(uint8_t, src2, [HCSCALE_FAST_SRC_SIZE + 32]);
+    LOCAL_ALIGNED_32(int16_t, dst0, [2048]);
+    LOCAL_ALIGNED_32(int16_t, dst1, [2048]);
+    LOCAL_ALIGNED_32(int16_t, dst2, [2048]);
+    LOCAL_ALIGNED_32(int16_t, dst3, [2048]);
+    SwsContext *sws;
+    SwsInternal *c;
+    int i, j;
+
+    declare_func(void, SwsInternal *c, int16_t *dst1, int16_t *dst2, int dstWidth,
+                 const uint8_t *src1, const uint8_t *src2, int srcW, int xInc);
+
+    sws = sws_alloc_context();
+    if (!sws || sws_init_context(sws, NULL, NULL) < 0)
+        fail();
+
+    c = sws_internal(sws);
+    c->srcBpc = 8;
+    c->dstBpc = 8;
+    c->opts.flags  = SWS_FAST_BILINEAR;
+    c->lumXInc = c->chrXInc = 83886;
+    ff_sws_init_scale(c);
+
+    if (check_func(c->hcscale_fast, "hcscale_fast")) {
+        for (i = 0; i < FF_ARRAY_ELEMS(dstW_list); i++) {
+            for (j = 0; j < FF_ARRAY_ELEMS(xInc_list); j++) {
+                int dstW = dstW_list[i];
+                int xInc = xInc_list[j];
+                int srcW = FFMIN(HCSCALE_FAST_SRC_SIZE,
+                                 (dstW * xInc) >> 16);
+
+                randomize_buffers(src1, srcW + 16);
+                randomize_buffers(src2, srcW + 16);
+                memset(dst0, 0, dstW * sizeof(dst0[0]));
+                memset(dst1, 0, dstW * sizeof(dst1[0]));
+                memset(dst2, 0, dstW * sizeof(dst2[0]));
+                memset(dst3, 0, dstW * sizeof(dst3[0]));
+
+                call_ref(NULL, dst0, dst2, dstW, src1, src2, srcW, xInc);
+                call_new(NULL, dst1, dst3, dstW, src1, src2, srcW, xInc);
+                if (memcmp(dst0, dst1, dstW * sizeof(dst0[0])) ||
+                    memcmp(dst2, dst3, dstW * sizeof(dst2[0])))
+                    fail();
+            }
+        }
+        bench_new(NULL, dst1, dst2, 300, src1, src2, 512, 83886);
+    }
+    sws_freeContext(sws);
+}
+#endif /* ARCH_LOONGARCH64 */
+
 void checkasm_check_sw_scale(void)
 {
     check_hscale();
     report("hscale");
+#if ARCH_LOONGARCH64
+    check_hyscale_fast();
+    check_hcscale_fast();
+    report("hscale_fast");
+#endif
     check_yuv2yuv1(0);
     check_yuv2yuv1(1);
     report("yuv2yuv1");