Commit b37f08942b for ffmpeg

commit b37f08942bb4f92d7a2525cd607bad1f43d88263
Author: Zhao Zhili <quinkblack@foxmail.com>
Date:   Tue Sep 15 11:45:31 2026 +0800

    avcodec/aarch64: add NEON HEVC SAO edge 10 and 12 bit

                                  Cortex-A510       Cortex-A715       Cortex-A725         Cortex-X3       Cortex-X925
    hevc_sao_edge_8_10_neon:    332.5 (4.07x)      56.9 (5.50x)      57.4 (4.64x)      37.0 (7.38x)      24.2 (7.96x)
    hevc_sao_edge_8_12_neon:    321.9 (4.22x)      57.1 (5.58x)      57.7 (4.70x)      37.0 (7.27x)      23.7 (8.12x)
    hevc_sao_edge_16_10_neon:  1280.4 (4.11x)     227.3 (5.30x)     227.9 (4.61x)     146.4 (7.27x)      94.9 (8.38x)
    hevc_sao_edge_16_12_neon:  1353.7 (3.96x)     227.9 (5.40x)     227.1 (4.63x)     146.4 (7.22x)      94.7 (8.40x)
    hevc_sao_edge_32_10_neon:  5249.9 (3.84x)     912.9 (5.15x)     911.5 (4.48x)     592.8 (7.21x)     372.7 (8.95x)
    hevc_sao_edge_32_12_neon:  5483.8 (3.67x)     912.6 (5.29x)     912.3 (4.47x)     579.8 (7.21x)     365.2 (9.11x)
    hevc_sao_edge_48_10_neon: 11147.5 (4.07x)    2035.5 (5.14x)    2033.9 (4.48x)    1324.3 (7.27x)     815.7 (9.29x)
    hevc_sao_edge_48_12_neon: 11179.6 (4.06x)    2035.5 (5.29x)    2035.9 (4.47x)    1324.9 (7.05x)     816.8 (9.25x)
    hevc_sao_edge_64_10_neon: 19965.6 (4.03x)    3606.1 (5.15x)    3603.6 (4.49x)    2340.0 (7.27x)    1440.0 (9.42x)
    hevc_sao_edge_64_12_neon: 20015.7 (4.02x)    3604.2 (5.30x)    3603.3 (4.49x)    2342.3 (7.07x)    1439.2 (9.40x)

    Signed-off-by: Zhao Zhili <zhilizhao@tencent.com>

diff --git a/libavcodec/aarch64/h26x/dsp.h b/libavcodec/aarch64/h26x/dsp.h
index 0cbbdc3157..dab5163c60 100644
--- a/libavcodec/aarch64/h26x/dsp.h
+++ b/libavcodec/aarch64/h26x/dsp.h
@@ -32,15 +32,21 @@ void ff_h26x_sao_band_filter_16x16_8_neon(uint8_t *_dst, const uint8_t *_src,
                                         ptrdiff_t stride_dst, ptrdiff_t stride_src,
                                         const int16_t *sao_offset_val, int sao_left_class,
                                         int width, int height);
-void ff_hevc_sao_edge_filter_16x16_8_neon(uint8_t *dst, const uint8_t *src, ptrdiff_t stride_dst,
-                                          const int16_t *sao_offset_val, int eo, int width, int height);
-void ff_hevc_sao_edge_filter_8x8_8_neon(uint8_t *dst, const uint8_t *src, ptrdiff_t stride_dst,
-                                        const int16_t *sao_offset_val, int eo, int width, int height);
-
-void ff_vvc_sao_edge_filter_16x16_8_neon(uint8_t *dst, const uint8_t *src, ptrdiff_t stride_dst,
-                                         const int16_t *sao_offset_val, int eo, int width, int height);
-void ff_vvc_sao_edge_filter_8x8_8_neon(uint8_t *dst, const uint8_t *src, ptrdiff_t stride_dst,
-                                       const int16_t *sao_offset_val, int eo, int width, int height);
+#define SAO_EDGE_FILTER_PROTO(codec, size, depth)                          \
+    void ff_##codec##_sao_edge_filter_##size##_##depth##_neon(             \
+        uint8_t *dst, const uint8_t *src, ptrdiff_t stride_dst,            \
+        const int16_t *sao_offset_val, int eo, int width, int height)
+
+SAO_EDGE_FILTER_PROTO(hevc, 16x16, 8);
+SAO_EDGE_FILTER_PROTO(hevc, 8x8,   8);
+SAO_EDGE_FILTER_PROTO(hevc, 16x16, 10);
+SAO_EDGE_FILTER_PROTO(hevc, 8x8,   10);
+SAO_EDGE_FILTER_PROTO(hevc, 16x16, 12);
+SAO_EDGE_FILTER_PROTO(hevc, 8x8,   12);
+SAO_EDGE_FILTER_PROTO(vvc,  16x16, 8);
+SAO_EDGE_FILTER_PROTO(vvc,  8x8,   8);
+
+#undef SAO_EDGE_FILTER_PROTO

 #define NEON8_FNPROTO_PARTIAL_6(fn, args, ext) \
     void ff_hevc_put_hevc_##fn##_h4_8_neon##ext args;  \
diff --git a/libavcodec/aarch64/h26x/sao_neon.S b/libavcodec/aarch64/h26x/sao_neon.S
index 90022fcfc7..d6a296db17 100644
--- a/libavcodec/aarch64/h26x/sao_neon.S
+++ b/libavcodec/aarch64/h26x/sao_neon.S
@@ -100,12 +100,18 @@ function ff_h26x_sao_band_filter_16x16_8_neon, export=1
         ret
 endfunc

-.Lhevc_sao_edge_pos:
+.Lhevc_sao_edge_pos_8:
 .word 1 // horizontal
 .word HEVC_SAO_STRIDE // vertical
 .word HEVC_SAO_STRIDE + 1 // 135 degree
 .word HEVC_SAO_STRIDE - 1 // 45 degree

+.Lhevc_sao_edge_pos_16bit:
+.word 2 // horizontal
+.word HEVC_SAO_STRIDE // vertical
+.word HEVC_SAO_STRIDE + 2 // 135 degree
+.word HEVC_SAO_STRIDE - 2 // 45 degree
+
 .Lvvc_sao_edge_pos:
 .word 1 // horizontal
 .word VVC_SAO_STRIDE // vertical
@@ -121,7 +127,7 @@ endfunc
 // ff_hevc_sao_edge_filter_16x16_8_neon(char *dst, char *src, ptrdiff stride_dst,
 //                                      int16 *sao_offset_val, int eo, int width, int height)
 function ff_hevc_sao_edge_filter_16x16_8_neon, export=1
-        adr             x7, .Lhevc_sao_edge_pos
+        adr             x7, .Lhevc_sao_edge_pos_8
         mov             x15, #HEVC_SAO_STRIDE
 1:
         ld1             {v3.8h}, [x3]              // load sao_offset_val
@@ -179,6 +185,84 @@ function ff_hevc_sao_edge_filter_16x16_8_neon, export=1
         ret
 endfunc

+/* Permutes the low byte of sao_offset_val into the byte table v28 by
+ * edge_idx = { 1, 2, 0, 3, 4 }.
+ * v29: bias 2
+ * v30: 0 for clip lower bound
+ * v31: clip higher bound, not set by this macro
+ */
+.macro  sao_edge_offsets_init
+        ld1             {v6.8h}, [x3]
+        mov             x11, #0x0402
+        movi            v29.8b, #2
+        movk            x11, #0x0600, lsl #16
+        movi            v30.8h, #0
+        movk            x11, #0x0008, lsl #32      // 0x0000000806000402
+        fmov            d7, x11
+        tbl             v28.8b, {v6.16b}, v7.8b
+.endm
+
+/* Filters one 8-pixel chunk
+ * v0: cur, and hold clipped result
+ * v1: a
+ * v2: b
+ */
+.macro  sao_edge_filter8
+        cmhi            v16.8h, v1.8h, v0.8h
+        cmhi            v17.8h, v0.8h, v1.8h
+        cmhi            v18.8h, v2.8h, v0.8h
+        cmhi            v19.8h, v0.8h, v2.8h
+        add             v20.8h, v16.8h, v18.8h
+        add             v21.8h, v17.8h, v19.8h
+        sub             v20.8h, v20.8h, v21.8h      // CMP(cur, a) + CMP(cur, b)
+        xtn             v20.8b, v20.8h
+        add             v20.8b, v20.8b, v29.8b      // offset table index
+        tbl             v16.8b, {v28.16b}, v20.8b
+        saddw           v0.8h, v0.8h, v16.8b
+        smax            v0.8h, v0.8h, v30.8h
+        smin            v0.8h, v0.8h, v31.8h
+.endm
+
+// ff_hevc_sao_edge_filter_16x16_12_neon(char *dst, char *src, ptrdiff stride_dst,
+//                                       int16 *sao_offset_val, int eo, int width, int height)
+function ff_hevc_sao_edge_filter_16x16_12_neon, export=1
+        mvni            v31.8h, #0xf0, lsl #8       // 4095
+        b               .Lhevc_sao_edge_16bit
+endfunc
+
+// ff_hevc_sao_edge_filter_16x16_10_neon(char *dst, char *src, ptrdiff stride_dst,
+//                                       int16 *sao_offset_val, int eo, int width, int height)
+function ff_hevc_sao_edge_filter_16x16_10_neon, export=1
+        mvni            v31.8h, #0xfc, lsl #8       // 1023
+.Lhevc_sao_edge_16bit:
+        adr             x7, .Lhevc_sao_edge_pos_16bit
+        mov             x15, #HEVC_SAO_STRIDE
+        add             w5, w5, #7
+        ldr             w4, [x7, w4, uxtw #2]       // a/b offsets in bytes
+        bic             w5, w5, #7
+        sao_edge_offsets_init
+        lsl             w5, w5, #1                  // width in bytes, multiple of 16
+        sub             x15, x15, x5                // src step to the next line
+        sub             x16, x2, x5                 // dst step to the next line
+1:
+        lsr             x14, x5, #4                 // 8-pixel groups
+        sub             x12, x1, x4
+        add             x13, x1, x4
+2:
+        ldr             q0, [x1], #16
+        ldr             q1, [x12], #16
+        ldr             q2, [x13], #16
+        subs            x14, x14, #1
+        sao_edge_filter8
+        str             q0, [x0], #16
+        b.ne            2b
+        subs            w6, w6, #1
+        add             x1, x1, x15
+        add             x0, x0, x16
+        b.ne            1b
+        ret
+endfunc
+
 function ff_vvc_sao_edge_filter_8x8_8_neon, export=1
         adr             x7, .Lvvc_sao_edge_pos
         mov             x15, #VVC_SAO_STRIDE
@@ -188,7 +272,7 @@ endfunc
 // ff_hevc_sao_edge_filter_8x8_8_neon(char *dst, char *src, ptrdiff stride_dst,
 //                                    int16 *sao_offset_val, int eo, int width, int height)
 function ff_hevc_sao_edge_filter_8x8_8_neon, export=1
-        adr             x7, .Lhevc_sao_edge_pos
+        adr             x7, .Lhevc_sao_edge_pos_8
         mov             x15, #HEVC_SAO_STRIDE
 1:
         ldr             w4, [x7, w4, uxtw #2]
@@ -235,3 +319,32 @@ function ff_hevc_sao_edge_filter_8x8_8_neon, export=1
         b.ne            1b
         ret
 endfunc
+
+// ff_hevc_sao_edge_filter_8x8_12_neon(char *dst, char *src, ptrdiff stride_dst,
+//                                     int16 *sao_offset_val, int eo, int width, int height)
+function ff_hevc_sao_edge_filter_8x8_12_neon, export=1
+        mvni            v31.8h, #0xf0, lsl #8     // 4095
+        b               .Lhevc_sao_edge_8x8_16bit
+endfunc
+
+// ff_hevc_sao_edge_filter_8x8_10_neon(char *dst, char *src, ptrdiff stride_dst,
+//                                     int16 *sao_offset_val, int eo, int width, int height)
+function ff_hevc_sao_edge_filter_8x8_10_neon, export=1
+        mvni            v31.8h, #0xfc, lsl #8     // 1023
+.Lhevc_sao_edge_8x8_16bit:
+        adr             x7, .Lhevc_sao_edge_pos_16bit
+        mov             x15, #HEVC_SAO_STRIDE
+        ldr             w4, [x7, w4, uxtw #2]      // a/b offsets in bytes
+        sao_edge_offsets_init
+        sub             x9,  x1, x4                // a neighbours
+        add             x10, x1, x4                // b neighbours
+1:
+        ld1             {v0.8h}, [x1], x15
+        ld1             {v1.8h}, [x9], x15
+        ld1             {v2.8h}, [x10], x15
+        subs            w6, w6, #1
+        sao_edge_filter8
+        st1             {v0.8h}, [x0], x2
+        b.ne            1b
+        ret
+endfunc
diff --git a/libavcodec/aarch64/hevcdsp_init_aarch64.c b/libavcodec/aarch64/hevcdsp_init_aarch64.c
index a2ca8aa124..12f8d93803 100644
--- a/libavcodec/aarch64/hevcdsp_init_aarch64.c
+++ b/libavcodec/aarch64/hevcdsp_init_aarch64.c
@@ -338,6 +338,11 @@ av_cold void ff_hevc_dsp_init_aarch64(HEVCDSPContext *c, const int bit_depth)
         c->idct_dc[2]                  = ff_hevc_idct_16x16_dc_10_neon;
         c->idct_dc[3]                  = ff_hevc_idct_32x32_dc_10_neon;
         c->dequant                     = hevc_dequant_10_neon;
+        c->sao_edge_filter[0]          = ff_hevc_sao_edge_filter_8x8_10_neon;
+        c->sao_edge_filter[1]          =
+        c->sao_edge_filter[2]          =
+        c->sao_edge_filter[3]          =
+        c->sao_edge_filter[4]          = ff_hevc_sao_edge_filter_16x16_10_neon;
     }
     if (bit_depth == 12) {
         c->hevc_h_loop_filter_luma     = ff_hevc_h_loop_filter_luma_12_neon;
@@ -353,5 +358,10 @@ av_cold void ff_hevc_dsp_init_aarch64(HEVCDSPContext *c, const int bit_depth)
         c->idct_dc[2]                  = ff_hevc_idct_16x16_dc_12_neon;
         c->idct_dc[3]                  = ff_hevc_idct_32x32_dc_12_neon;
         c->dequant                     = hevc_dequant_12_neon;
+        c->sao_edge_filter[0]          = ff_hevc_sao_edge_filter_8x8_12_neon;
+        c->sao_edge_filter[1]          =
+        c->sao_edge_filter[2]          =
+        c->sao_edge_filter[3]          =
+        c->sao_edge_filter[4]          = ff_hevc_sao_edge_filter_16x16_12_neon;
     }
 }