Commit b91a82d6dd for ffmpeg
commit b91a82d6dd0e7c32caf34b345130c8fb1309a597
Author: jinbo <jinbo@loongson.cn>
Date: Tue Sep 1 11:28:33 2026 +0800
swscale/loongarch: add LSX fast bilinear horizontal scale
Add LoongArch LSX optimized implementations of the fast bilinear
horizontal scaler (SWS_FAST_BILINEAR, hyscale_fast/hcscale_fast).
The vectorized loop processes 8 destination pixels per iteration,
computing the source offsets and interpolation weights in registers
and gathering the input samples with vshuf_b over a 32-byte window.
It covers scaling ratios down to 4x downscaling; beyond that the
scalar loop is used.
checkasm --bench on 3A5000 4 cores 2.5GHz:
hcscale_fast_c: 99.7
hcscale_fast_lsx: 28.1 ( 3.55x)
hyscale_fast_c: 51.1
hyscale_fast_lsx: 18.3 ( 2.78x)
Performance with:
$ ./ffmpeg -cpuflags lsx -f lavfi -i "smptebars=size=3840x2160:rate=30:duration=100,format=nv12" \
-threads 1 -vf "scale=1920:1080:sws_flags=fast_bilinear,format=bgra" -f null -
before: 36fps
after : 86fps
diff --git a/libswscale/loongarch/Makefile b/libswscale/loongarch/Makefile
index 06aed9d245..0b9c87100e 100644
--- a/libswscale/loongarch/Makefile
+++ b/libswscale/loongarch/Makefile
@@ -6,6 +6,7 @@ LASX-OBJS-$(CONFIG_SWSCALE) += loongarch/swscale_lasx.o \
loongarch/output_lasx.o
LSX-OBJS-$(CONFIG_SWSCALE) += loongarch/swscale.o \
loongarch/swscale_unscaled.o \
+ loongarch/hscale_fast_bilinear_lsx.o \
loongarch/swscale_lsx.o \
loongarch/input.o \
loongarch/output.o \
diff --git a/libswscale/loongarch/hscale_fast_bilinear_lsx.c b/libswscale/loongarch/hscale_fast_bilinear_lsx.c
new file mode 100644
index 0000000000..7896542251
--- /dev/null
+++ b/libswscale/loongarch/hscale_fast_bilinear_lsx.c
@@ -0,0 +1,162 @@
+/*
+ * Copyright (C) 2026 Loongson Technology Co. Ltd.
+ * Contributed by Bo Jin(jinbo@loongson.cn)
+ *
+ * This file is part of FFmpeg.
+ *
+ * FFmpeg is free software; you can redistribute it and/or
+ * modify it under the terms of the GNU Lesser General Public
+ * License as published by the Free Software Foundation; either
+ * version 2.1 of the License, or (at your option) any later version.
+ *
+ * FFmpeg is distributed in the hope that it will be useful,
+ * but WITHOUT ANY WARRANTY; without even the implied warranty of
+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
+ * Lesser General Public License for more details.
+ *
+ * You should have received a copy of the GNU Lesser General Public
+ * License along with FFmpeg; if not, write to the Free Software
+ * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
+ */
+
+#include "swscale_loongarch.h"
+#include "libavutil/loongarch/loongson_intrinsics.h"
+
+/* Each vector iteration scales 8 destination pixels. Their source
+ * position offsets ((xpos & 0xFFFF) + j*xInc) >> 16, j in [0, 7], must
+ * stay within the 32-byte gather window, which holds while
+ * xInc <= (1 << 18); otherwise fall back to the scalar loop. */
+#define LSX_HSCALE_FAST_MAX_XINC (1 << 18)
+
+void ff_hyscale_fast_lsx(SwsInternal *c, int16_t *dst, int dstWidth,
+ const uint8_t *src, int srcW, int xInc)
+{
+ int i = 0;
+ unsigned int xpos = 0;
+
+ if (xInc <= LSX_HSCALE_FAST_MAX_XINC) {
+ static const int32_t idx32[4] = {0, 1, 2, 3};
+ static const int16_t idx16[8] = {0, 1, 2, 3, 4, 5, 6, 7};
+ static const uint8_t shuf8[16] = {
+ 0, 4, 8, 12, 16, 20, 24, 28,
+ 0x30, 0x30, 0x30, 0x30, 0x30, 0x30, 0x30, 0x30
+ };
+
+ /* [0, xInc, 2*xInc, 3*xInc] as 32-bit lanes */
+ __m128i vadd_w = __lsx_vmul_w(__lsx_vreplgr2vr_w(xInc),
+ __lsx_vld(idx32, 0));
+ /* [0, xInc, ..., 7*xInc] modulo 2^16 as 16-bit lanes */
+ __m128i vadd16 = __lsx_vmul_h(__lsx_vreplgr2vr_h(xInc),
+ __lsx_vld(idx16, 0));
+ __m128i vx4 = __lsx_vreplgr2vr_w(4 * xInc);
+ __m128i vshuf8 = __lsx_vld(shuf8, 0);
+ __m128i v128 = __lsx_vreplgr2vr_h(128);
+
+ for (; i + 8 <= dstWidth && (xpos >> 16) + 32 < srcW; i += 8, xpos += 8 * xInc) {
+ unsigned int lo = xpos & 0xFFFF;
+ unsigned int xx = xpos >> 16;
+
+ /* full 32-bit positions of the 8 pixels */
+ __m128i vc0 = __lsx_vadd_w(__lsx_vreplgr2vr_w(lo), vadd_w);
+ __m128i vc1 = __lsx_vadd_w(vc0, vx4);
+
+ /* source offsets (j = 0..7), each in [0, 28] */
+ __m128i vperm = __lsx_vshuf_b(__lsx_vsrli_w(vc1, 16),
+ __lsx_vsrli_w(vc0, 16), vshuf8);
+
+ /* xalpha = (xpos & 0xFFFF) >> 9, 16-bit lanes */
+ __m128i valpha = __lsx_vsrli_h(__lsx_vadd_h(__lsx_vreplgr2vr_h(lo),
+ vadd16), 9);
+
+ __m128i v0 = __lsx_vsllwil_hu_bu(__lsx_vshuf_b(__lsx_vld(src + xx + 16, 0),
+ __lsx_vld(src + xx, 0),
+ vperm), 0);
+ __m128i v1 = __lsx_vsllwil_hu_bu(__lsx_vshuf_b(__lsx_vld(src + xx + 17, 0),
+ __lsx_vld(src + xx + 1, 0),
+ vperm), 0);
+
+ __m128i w0 = __lsx_vsub_h(v128, valpha);
+ __lsx_vst(__lsx_vadd_h(__lsx_vmul_h(v0, w0), __lsx_vmul_h(v1, valpha)),
+ dst + i, 0);
+ }
+ }
+
+ for (; i < dstWidth; i++) {
+ unsigned int xx = xpos >> 16;
+ unsigned int xalpha = (xpos & 0xFFFF) >> 9;
+ dst[i] = (src[xx] << 7) + (src[xx + 1] - src[xx]) * xalpha;
+ xpos += xInc;
+ }
+ for (i = dstWidth - 1; (i * (int64_t)xInc) >> 16 >= srcW - 1; i--)
+ dst[i] = src[srcW - 1] * 128;
+}
+
+void ff_hcscale_fast_lsx(SwsInternal *c, int16_t *dst1, int16_t *dst2,
+ int dstWidth, const uint8_t *src1,
+ const uint8_t *src2, int srcW, int xInc)
+{
+ int i = 0;
+ unsigned int xpos = 0;
+
+ if (xInc <= LSX_HSCALE_FAST_MAX_XINC) {
+ static const int32_t idx32[4] = {0, 1, 2, 3};
+ static const int16_t idx16[8] = {0, 1, 2, 3, 4, 5, 6, 7};
+ static const uint8_t shuf8[16] = {
+ 0, 4, 8, 12, 16, 20, 24, 28,
+ 0x30, 0x30, 0x30, 0x30, 0x30, 0x30, 0x30, 0x30
+ };
+
+ __m128i vadd_w = __lsx_vmul_w(__lsx_vreplgr2vr_w(xInc),
+ __lsx_vld(idx32, 0));
+ __m128i vadd16 = __lsx_vmul_h(__lsx_vreplgr2vr_h(xInc),
+ __lsx_vld(idx16, 0));
+ __m128i vx4 = __lsx_vreplgr2vr_w(4 * xInc);
+ __m128i vshuf8 = __lsx_vld(shuf8, 0);
+ __m128i v127 = __lsx_vreplgr2vr_h(127);
+
+ for (; i + 8 <= dstWidth && (xpos >> 16) + 32 < srcW; i += 8, xpos += 8 * xInc) {
+ unsigned int lo = xpos & 0xFFFF;
+ unsigned int xx = xpos >> 16;
+
+ __m128i vc0 = __lsx_vadd_w(__lsx_vreplgr2vr_w(lo), vadd_w);
+ __m128i vc1 = __lsx_vadd_w(vc0, vx4);
+
+ __m128i vperm = __lsx_vshuf_b(__lsx_vsrli_w(vc1, 16),
+ __lsx_vsrli_w(vc0, 16), vshuf8);
+
+ __m128i valpha = __lsx_vsrli_h(__lsx_vadd_h(__lsx_vreplgr2vr_h(lo),
+ vadd16), 9);
+
+ __m128i v10 = __lsx_vsllwil_hu_bu(__lsx_vshuf_b(__lsx_vld(src1 + xx + 16, 0),
+ __lsx_vld(src1 + xx, 0),
+ vperm), 0);
+ __m128i v11 = __lsx_vsllwil_hu_bu(__lsx_vshuf_b(__lsx_vld(src1 + xx + 17, 0),
+ __lsx_vld(src1 + xx + 1, 0),
+ vperm), 0);
+ __m128i v20 = __lsx_vsllwil_hu_bu(__lsx_vshuf_b(__lsx_vld(src2 + xx + 16, 0),
+ __lsx_vld(src2 + xx, 0),
+ vperm), 0);
+ __m128i v21 = __lsx_vsllwil_hu_bu(__lsx_vshuf_b(__lsx_vld(src2 + xx + 17, 0),
+ __lsx_vld(src2 + xx + 1, 0),
+ vperm), 0);
+
+ __m128i w0 = __lsx_vsub_h(v127, valpha);
+ __lsx_vst(__lsx_vadd_h(__lsx_vmul_h(v10, w0), __lsx_vmul_h(v11, valpha)),
+ dst1 + i, 0);
+ __lsx_vst(__lsx_vadd_h(__lsx_vmul_h(v20, w0), __lsx_vmul_h(v21, valpha)),
+ dst2 + i, 0);
+ }
+ }
+
+ for (; i < dstWidth; i++) {
+ unsigned int xx = xpos >> 16;
+ unsigned int xalpha = (xpos & 0xFFFF) >> 9;
+ dst1[i] = (src1[xx] * (xalpha ^ 127) + src1[xx + 1] * xalpha);
+ dst2[i] = (src2[xx] * (xalpha ^ 127) + src2[xx + 1] * xalpha);
+ xpos += xInc;
+ }
+ for (i = dstWidth - 1; (i * (int64_t)xInc) >> 16 >= srcW - 1; i--) {
+ dst1[i] = src1[srcW - 1] * 128;
+ dst2[i] = src2[srcW - 1] * 128;
+ }
+}
diff --git a/libswscale/loongarch/swscale_init_loongarch.c b/libswscale/loongarch/swscale_init_loongarch.c
index 0c937b047f..ced73f2b0f 100644
--- a/libswscale/loongarch/swscale_init_loongarch.c
+++ b/libswscale/loongarch/swscale_init_loongarch.c
@@ -71,6 +71,12 @@ av_cold void ff_sws_init_swscale_loongarch(SwsInternal *c)
c->hyScale = c->hcScale = c->dstBpc > 14 ? ff_hscale_16_to_19_lsx
: ff_hscale_16_to_15_lsx;
}
+ if (c->srcBpc == 8 && c->dstBpc <= 14 &&
+ c->opts.flags & SWS_FAST_BILINEAR &&
+ c->lumXInc <= (1 << 18) && c->chrXInc <= (1 << 18)) {
+ c->hyscale_fast = ff_hyscale_fast_lsx;
+ c->hcscale_fast = ff_hcscale_fast_lsx;
+ }
}
#if HAVE_LASX
if (have_lasx(cpu_flags)) {
diff --git a/libswscale/loongarch/swscale_loongarch.h b/libswscale/loongarch/swscale_loongarch.h
index a8297b6972..b2875ba70f 100644
--- a/libswscale/loongarch/swscale_loongarch.h
+++ b/libswscale/loongarch/swscale_loongarch.h
@@ -50,6 +50,13 @@ void ff_hscale_16_to_19_sub_lsx(SwsInternal *c, int16_t *_dst, int dstW,
const uint8_t *_src, const int16_t *filter,
const int32_t *filterPos, int filterSize, int sh);
+void ff_hyscale_fast_lsx(SwsInternal *c, int16_t *dst, int dstWidth,
+ const uint8_t *src, int srcW, int xInc);
+
+void ff_hcscale_fast_lsx(SwsInternal *c, int16_t *dst1, int16_t *dst2,
+ int dstWidth, const uint8_t *src1,
+ const uint8_t *src2, int srcW, int xInc);
+
void lumRangeFromJpeg_lsx(int16_t *dst, int width, uint32_t coeff, int64_t offset);
void chrRangeFromJpeg_lsx(int16_t *dstU, int16_t *dstV, int width, uint32_t coeff, int64_t offset);
void lumRangeToJpeg_lsx(int16_t *dst, int width, uint32_t coeff, int64_t offset);
diff --git a/tests/checkasm/sw_scale.c b/tests/checkasm/sw_scale.c
index b06ac23392..68d8f8d735 100644
--- a/tests/checkasm/sw_scale.c
+++ b/tests/checkasm/sw_scale.c
@@ -492,10 +492,130 @@ static void check_hscale(void)
sws_freeContext(sws);
}
+/* Fast-bilinear horizontal scaling (SWS_FAST_BILINEAR, c->hyscale_fast /
+ * c->hcscale_fast): the optimized implementation must be bit-exact with
+ * the C reference. Catches e.g. swapped gather-position/weight tables in
+ * the LoongArch LSX implementation, which no FATE frame test covers.
+ *
+ * Only enabled on LoongArch: other archs' fast-bilinear implementations
+ * are not guaranteed bit-exact with the C reference (e.g. x86 MMXEXT uses
+ * complementary weights), so a cross-arch comparison would fail there. */
+#if ARCH_LOONGARCH64
+static void check_hyscale_fast(void)
+{
+#define HSCALE_FAST_SRC_SIZE 4096
+ static const int dstW_list[] = { 8, 16, 63, 100, 255, 256, 300, 511, 512, 1024 };
+ static const int xInc_list[] = { 32768, 65536, 83886, 98304, 131072, 200000, 262143, 262144 };
+ LOCAL_ALIGNED_32(uint8_t, src, [HSCALE_FAST_SRC_SIZE + 32]);
+ LOCAL_ALIGNED_32(int16_t, dst0, [2048]);
+ LOCAL_ALIGNED_32(int16_t, dst1, [2048]);
+ SwsContext *sws;
+ SwsInternal *c;
+ int i, j;
+
+ declare_func(void, SwsInternal *c, int16_t *dst, int dstWidth,
+ const uint8_t *src, int srcW, int xInc);
+
+ sws = sws_alloc_context();
+ if (!sws || sws_init_context(sws, NULL, NULL) < 0)
+ fail();
+
+ c = sws_internal(sws);
+ c->srcBpc = 8;
+ c->dstBpc = 8;
+ c->opts.flags = SWS_FAST_BILINEAR;
+ c->lumXInc = c->chrXInc = 83886;
+ ff_sws_init_scale(c);
+
+ if (check_func(c->hyscale_fast, "hyscale_fast")) {
+ for (i = 0; i < FF_ARRAY_ELEMS(dstW_list); i++) {
+ for (j = 0; j < FF_ARRAY_ELEMS(xInc_list); j++) {
+ int dstW = dstW_list[i];
+ int xInc = xInc_list[j];
+ int srcW = FFMIN(HSCALE_FAST_SRC_SIZE,
+ (dstW * xInc) >> 16);
+
+ randomize_buffers(src, srcW + 16);
+ memset(dst0, 0, dstW * sizeof(dst0[0]));
+ memset(dst1, 0, dstW * sizeof(dst1[0]));
+
+ call_ref(NULL, dst0, dstW, src, srcW, xInc);
+ call_new(NULL, dst1, dstW, src, srcW, xInc);
+ if (memcmp(dst0, dst1, dstW * sizeof(dst0[0])))
+ fail();
+ }
+ }
+ bench_new(NULL, dst1, 300, src, 512, 83886);
+ }
+ sws_freeContext(sws);
+}
+
+static void check_hcscale_fast(void)
+{
+#define HCSCALE_FAST_SRC_SIZE 4096
+ static const int dstW_list[] = { 8, 16, 63, 100, 255, 256, 300, 511, 512, 1024 };
+ static const int xInc_list[] = { 32768, 65536, 83886, 98304, 131072, 200000, 262143, 262144 };
+ LOCAL_ALIGNED_32(uint8_t, src1, [HCSCALE_FAST_SRC_SIZE + 32]);
+ LOCAL_ALIGNED_32(uint8_t, src2, [HCSCALE_FAST_SRC_SIZE + 32]);
+ LOCAL_ALIGNED_32(int16_t, dst0, [2048]);
+ LOCAL_ALIGNED_32(int16_t, dst1, [2048]);
+ LOCAL_ALIGNED_32(int16_t, dst2, [2048]);
+ LOCAL_ALIGNED_32(int16_t, dst3, [2048]);
+ SwsContext *sws;
+ SwsInternal *c;
+ int i, j;
+
+ declare_func(void, SwsInternal *c, int16_t *dst1, int16_t *dst2, int dstWidth,
+ const uint8_t *src1, const uint8_t *src2, int srcW, int xInc);
+
+ sws = sws_alloc_context();
+ if (!sws || sws_init_context(sws, NULL, NULL) < 0)
+ fail();
+
+ c = sws_internal(sws);
+ c->srcBpc = 8;
+ c->dstBpc = 8;
+ c->opts.flags = SWS_FAST_BILINEAR;
+ c->lumXInc = c->chrXInc = 83886;
+ ff_sws_init_scale(c);
+
+ if (check_func(c->hcscale_fast, "hcscale_fast")) {
+ for (i = 0; i < FF_ARRAY_ELEMS(dstW_list); i++) {
+ for (j = 0; j < FF_ARRAY_ELEMS(xInc_list); j++) {
+ int dstW = dstW_list[i];
+ int xInc = xInc_list[j];
+ int srcW = FFMIN(HCSCALE_FAST_SRC_SIZE,
+ (dstW * xInc) >> 16);
+
+ randomize_buffers(src1, srcW + 16);
+ randomize_buffers(src2, srcW + 16);
+ memset(dst0, 0, dstW * sizeof(dst0[0]));
+ memset(dst1, 0, dstW * sizeof(dst1[0]));
+ memset(dst2, 0, dstW * sizeof(dst2[0]));
+ memset(dst3, 0, dstW * sizeof(dst3[0]));
+
+ call_ref(NULL, dst0, dst2, dstW, src1, src2, srcW, xInc);
+ call_new(NULL, dst1, dst3, dstW, src1, src2, srcW, xInc);
+ if (memcmp(dst0, dst1, dstW * sizeof(dst0[0])) ||
+ memcmp(dst2, dst3, dstW * sizeof(dst2[0])))
+ fail();
+ }
+ }
+ bench_new(NULL, dst1, dst2, 300, src1, src2, 512, 83886);
+ }
+ sws_freeContext(sws);
+}
+#endif /* ARCH_LOONGARCH64 */
+
void checkasm_check_sw_scale(void)
{
check_hscale();
report("hscale");
+#if ARCH_LOONGARCH64
+ check_hyscale_fast();
+ check_hcscale_fast();
+ report("hscale_fast");
+#endif
check_yuv2yuv1(0);
check_yuv2yuv1(1);
report("yuv2yuv1");