Commit 26c6816846 for ffmpeg

commit 26c68168461d57519b18fbceac92a58fe7ad1db6
Author: Zhao Zhili <quinkblack@foxmail.com>
Date:   Tue Sep 15 11:46:01 2026 +0800

    avcodec/aarch64: add NEON VVC SAO edge 10 and 12 bit

                                  Cortex-A510       Cortex-A715       Cortex-A725         Cortex-X3       Cortex-X925
    vvc_sao_edge_8_10_neon:     308.9 (4.29x)      57.1 (5.47x)      57.4 (4.66x)      36.4 (7.35x)      23.7 (8.13x)
    vvc_sao_edge_8_12_neon:     314.6 (4.19x)      57.3 (5.58x)      58.3 (4.59x)      36.4 (7.34x)      23.9 (8.08x)
    vvc_sao_edge_16_10_neon:   1202.1 (4.28x)     228.1 (5.26x)     228.6 (4.54x)     144.6 (7.28x)      95.0 (8.39x)
    vvc_sao_edge_16_12_neon:   1203.3 (4.28x)     229.4 (5.37x)     239.4 (4.33x)     144.6 (7.22x)      95.0 (8.38x)
    vvc_sao_edge_32_10_neon:   4718.8 (4.28x)     913.1 (5.12x)     911.5 (4.42x)     571.5 (7.30x)     366.9 (9.08x)
    vvc_sao_edge_32_12_neon:   4737.9 (4.26x)     913.3 (5.28x)     913.0 (4.41x)     571.8 (7.31x)     365.6 (9.15x)
    vvc_sao_edge_48_10_neon:  10970.3 (4.16x)    2035.0 (5.13x)    2036.2 (4.43x)    1307.6 (7.14x)     823.6 (9.19x)
    vvc_sao_edge_48_12_neon:  10980.7 (4.15x)    2037.3 (5.29x)    2035.0 (4.43x)    1311.2 (7.14x)     824.1 (9.20x)
    vvc_sao_edge_64_10_neon:  19127.9 (4.23x)    3605.3 (5.13x)    3628.5 (4.41x)    2315.9 (7.17x)    1452.7 (9.33x)
    vvc_sao_edge_64_12_neon:  19136.2 (4.23x)    3606.3 (5.29x)    3629.0 (4.40x)    2320.9 (7.17x)    1448.9 (9.36x)
    vvc_sao_edge_80_10_neon:  29560.9 (4.29x)    5624.0 (5.30x)    5657.5 (4.64x)    3610.8 (7.17x)    2252.1 (9.44x)
    vvc_sao_edge_80_12_neon:  29644.8 (4.28x)    5626.8 (5.48x)    5661.2 (4.64x)    3614.4 (7.18x)    2250.3 (9.46x)
    vvc_sao_edge_96_10_neon:  42316.0 (4.29x)    8088.8 (5.28x)    8135.3 (4.61x)    5188.5 (7.20x)    3236.3 (9.49x)
    vvc_sao_edge_96_12_neon:  42863.0 (4.23x)    8091.9 (5.45x)    8136.9 (4.61x)    5190.6 (7.21x)    3236.1 (9.48x)
    vvc_sao_edge_112_10_neon: 57410.2 (4.30x)   11001.8 (5.27x)   11056.2 (4.59x)    7066.3 (7.19x)    4407.8 (9.48x)
    vvc_sao_edge_112_12_neon: 57372.5 (4.30x)   11006.7 (5.44x)   11054.2 (4.59x)    7074.3 (7.19x)    4400.6 (9.52x)
    vvc_sao_edge_128_10_neon: 75737.0 (4.26x)   14401.4 (5.26x)   14468.5 (4.57x)    9227.8 (7.22x)    5729.6 (9.54x)
    vvc_sao_edge_128_12_neon: 75824.1 (4.25x)   14404.0 (5.42x)   14472.1 (4.57x)    9222.1 (7.20x)    5730.8 (9.55x)

    Signed-off-by: Zhao Zhili <zhilizhao@tencent.com>

diff --git a/libavcodec/aarch64/h26x/dsp.h b/libavcodec/aarch64/h26x/dsp.h
index dab5163c60..2b8eae130a 100644
--- a/libavcodec/aarch64/h26x/dsp.h
+++ b/libavcodec/aarch64/h26x/dsp.h
@@ -45,6 +45,10 @@ SAO_EDGE_FILTER_PROTO(hevc, 16x16, 12);
 SAO_EDGE_FILTER_PROTO(hevc, 8x8,   12);
 SAO_EDGE_FILTER_PROTO(vvc,  16x16, 8);
 SAO_EDGE_FILTER_PROTO(vvc,  8x8,   8);
+SAO_EDGE_FILTER_PROTO(vvc,  16x16, 10);
+SAO_EDGE_FILTER_PROTO(vvc,  8x8,   10);
+SAO_EDGE_FILTER_PROTO(vvc,  16x16, 12);
+SAO_EDGE_FILTER_PROTO(vvc,  8x8,   12);

 #undef SAO_EDGE_FILTER_PROTO

diff --git a/libavcodec/aarch64/h26x/sao_neon.S b/libavcodec/aarch64/h26x/sao_neon.S
index d6a296db17..083f0eb89d 100644
--- a/libavcodec/aarch64/h26x/sao_neon.S
+++ b/libavcodec/aarch64/h26x/sao_neon.S
@@ -118,6 +118,12 @@ endfunc
 .word VVC_SAO_STRIDE + 1 // 135 degree
 .word VVC_SAO_STRIDE - 1 // 45 degree

+.Lvvc_sao_edge_pos_16bit:
+.word 2 // horizontal
+.word VVC_SAO_STRIDE // vertical
+.word VVC_SAO_STRIDE + 2 // 135 degree
+.word VVC_SAO_STRIDE - 2 // 45 degree
+
 function ff_vvc_sao_edge_filter_16x16_8_neon, export=1
         adr             x7, .Lvvc_sao_edge_pos
         mov             x15, #VVC_SAO_STRIDE
@@ -223,6 +229,23 @@ endfunc
         smin            v0.8h, v0.8h, v31.8h
 .endm

+// ff_vvc_sao_edge_filter_16x16_12_neon(char *dst, char *src, ptrdiff stride_dst,
+//                                      int16 *sao_offset_val, int eo, int width, int height)
+function ff_vvc_sao_edge_filter_16x16_12_neon, export=1
+        mvni            v31.8h, #0xf0, lsl #8       // 4095
+        b               .Lvvc_sao_edge_16x16_16bit
+endfunc
+
+// ff_vvc_sao_edge_filter_16x16_10_neon(char *dst, char *src, ptrdiff stride_dst,
+//                                      int16 *sao_offset_val, int eo, int width, int height)
+function ff_vvc_sao_edge_filter_16x16_10_neon, export=1
+        mvni            v31.8h, #0xfc, lsl #8       // 1023
+.Lvvc_sao_edge_16x16_16bit:
+        adr             x7, .Lvvc_sao_edge_pos_16bit
+        mov             x15, #VVC_SAO_STRIDE
+        b               .Lh26x_sao_edge_16x16_16bit
+endfunc
+
 // ff_hevc_sao_edge_filter_16x16_12_neon(char *dst, char *src, ptrdiff stride_dst,
 //                                       int16 *sao_offset_val, int eo, int width, int height)
 function ff_hevc_sao_edge_filter_16x16_12_neon, export=1
@@ -237,6 +260,7 @@ function ff_hevc_sao_edge_filter_16x16_10_neon, export=1
 .Lhevc_sao_edge_16bit:
         adr             x7, .Lhevc_sao_edge_pos_16bit
         mov             x15, #HEVC_SAO_STRIDE
+.Lh26x_sao_edge_16x16_16bit:
         add             w5, w5, #7
         ldr             w4, [x7, w4, uxtw #2]       // a/b offsets in bytes
         bic             w5, w5, #7
@@ -320,6 +344,23 @@ function ff_hevc_sao_edge_filter_8x8_8_neon, export=1
         ret
 endfunc

+// ff_vvc_sao_edge_filter_8x8_12_neon(char *dst, char *src, ptrdiff stride_dst,
+//                                    int16 *sao_offset_val, int eo, int width, int height)
+function ff_vvc_sao_edge_filter_8x8_12_neon, export=1
+        mvni            v31.8h, #0xf0, lsl #8       // 4095
+        b               .Lvvc_sao_edge_8x8_16bit
+endfunc
+
+// ff_vvc_sao_edge_filter_8x8_10_neon(char *dst, char *src, ptrdiff stride_dst,
+//                                    int16 *sao_offset_val, int eo, int width, int height)
+function ff_vvc_sao_edge_filter_8x8_10_neon, export=1
+        mvni            v31.8h, #0xfc, lsl #8       // 1023
+.Lvvc_sao_edge_8x8_16bit:
+        adr             x7, .Lvvc_sao_edge_pos_16bit
+        mov             x15, #VVC_SAO_STRIDE
+        b               .Lh26x_sao_edge_8x8_16bit
+endfunc
+
 // ff_hevc_sao_edge_filter_8x8_12_neon(char *dst, char *src, ptrdiff stride_dst,
 //                                     int16 *sao_offset_val, int eo, int width, int height)
 function ff_hevc_sao_edge_filter_8x8_12_neon, export=1
@@ -334,6 +375,7 @@ function ff_hevc_sao_edge_filter_8x8_10_neon, export=1
 .Lhevc_sao_edge_8x8_16bit:
         adr             x7, .Lhevc_sao_edge_pos_16bit
         mov             x15, #HEVC_SAO_STRIDE
+.Lh26x_sao_edge_8x8_16bit:
         ldr             w4, [x7, w4, uxtw #2]      // a/b offsets in bytes
         sao_edge_offsets_init
         sub             x9,  x1, x4                // a neighbours
diff --git a/libavcodec/aarch64/vvc/dsp_init.c b/libavcodec/aarch64/vvc/dsp_init.c
index 53bf9f9edd..8936a4606a 100644
--- a/libavcodec/aarch64/vvc/dsp_init.c
+++ b/libavcodec/aarch64/vvc/dsp_init.c
@@ -372,6 +372,10 @@ void ff_vvc_dsp_init_aarch64(VVCDSPContext *const c, const int bd)
         c->inter.put[1][5][1][1] =
         c->inter.put[1][6][1][1] = ff_vvc_put_chroma_hv_x16_10_neon;

+        c->sao.edge_filter[0] = ff_vvc_sao_edge_filter_8x8_10_neon;
+        for (int i = 1; i < FF_ARRAY_ELEMS(c->sao.edge_filter); i++)
+            c->sao.edge_filter[i] = ff_vvc_sao_edge_filter_16x16_10_neon;
+
         c->alf.filter[LUMA] = alf_filter_luma_10_neon;
         c->alf.filter[CHROMA] = alf_filter_chroma_10_neon;
         c->alf.classify = alf_classify_10_neon;
@@ -424,6 +428,10 @@ void ff_vvc_dsp_init_aarch64(VVCDSPContext *const c, const int bd)
         c->inter.put[1][5][1][1] =
         c->inter.put[1][6][1][1] = ff_vvc_put_chroma_hv_x16_12_neon;

+        c->sao.edge_filter[0] = ff_vvc_sao_edge_filter_8x8_12_neon;
+        for (int i = 1; i < FF_ARRAY_ELEMS(c->sao.edge_filter); i++)
+            c->sao.edge_filter[i] = ff_vvc_sao_edge_filter_16x16_12_neon;
+
         c->alf.filter[LUMA] = alf_filter_luma_12_neon;
         c->alf.filter[CHROMA] = alf_filter_chroma_12_neon;
         c->alf.classify = alf_classify_12_neon;