Commit b37f08942b for ffmpeg
commit b37f08942bb4f92d7a2525cd607bad1f43d88263
Author: Zhao Zhili <quinkblack@foxmail.com>
Date: Tue Sep 15 11:45:31 2026 +0800
avcodec/aarch64: add NEON HEVC SAO edge 10 and 12 bit
Cortex-A510 Cortex-A715 Cortex-A725 Cortex-X3 Cortex-X925
hevc_sao_edge_8_10_neon: 332.5 (4.07x) 56.9 (5.50x) 57.4 (4.64x) 37.0 (7.38x) 24.2 (7.96x)
hevc_sao_edge_8_12_neon: 321.9 (4.22x) 57.1 (5.58x) 57.7 (4.70x) 37.0 (7.27x) 23.7 (8.12x)
hevc_sao_edge_16_10_neon: 1280.4 (4.11x) 227.3 (5.30x) 227.9 (4.61x) 146.4 (7.27x) 94.9 (8.38x)
hevc_sao_edge_16_12_neon: 1353.7 (3.96x) 227.9 (5.40x) 227.1 (4.63x) 146.4 (7.22x) 94.7 (8.40x)
hevc_sao_edge_32_10_neon: 5249.9 (3.84x) 912.9 (5.15x) 911.5 (4.48x) 592.8 (7.21x) 372.7 (8.95x)
hevc_sao_edge_32_12_neon: 5483.8 (3.67x) 912.6 (5.29x) 912.3 (4.47x) 579.8 (7.21x) 365.2 (9.11x)
hevc_sao_edge_48_10_neon: 11147.5 (4.07x) 2035.5 (5.14x) 2033.9 (4.48x) 1324.3 (7.27x) 815.7 (9.29x)
hevc_sao_edge_48_12_neon: 11179.6 (4.06x) 2035.5 (5.29x) 2035.9 (4.47x) 1324.9 (7.05x) 816.8 (9.25x)
hevc_sao_edge_64_10_neon: 19965.6 (4.03x) 3606.1 (5.15x) 3603.6 (4.49x) 2340.0 (7.27x) 1440.0 (9.42x)
hevc_sao_edge_64_12_neon: 20015.7 (4.02x) 3604.2 (5.30x) 3603.3 (4.49x) 2342.3 (7.07x) 1439.2 (9.40x)
Signed-off-by: Zhao Zhili <zhilizhao@tencent.com>
diff --git a/libavcodec/aarch64/h26x/dsp.h b/libavcodec/aarch64/h26x/dsp.h
index 0cbbdc3157..dab5163c60 100644
--- a/libavcodec/aarch64/h26x/dsp.h
+++ b/libavcodec/aarch64/h26x/dsp.h
@@ -32,15 +32,21 @@ void ff_h26x_sao_band_filter_16x16_8_neon(uint8_t *_dst, const uint8_t *_src,
ptrdiff_t stride_dst, ptrdiff_t stride_src,
const int16_t *sao_offset_val, int sao_left_class,
int width, int height);
-void ff_hevc_sao_edge_filter_16x16_8_neon(uint8_t *dst, const uint8_t *src, ptrdiff_t stride_dst,
- const int16_t *sao_offset_val, int eo, int width, int height);
-void ff_hevc_sao_edge_filter_8x8_8_neon(uint8_t *dst, const uint8_t *src, ptrdiff_t stride_dst,
- const int16_t *sao_offset_val, int eo, int width, int height);
-
-void ff_vvc_sao_edge_filter_16x16_8_neon(uint8_t *dst, const uint8_t *src, ptrdiff_t stride_dst,
- const int16_t *sao_offset_val, int eo, int width, int height);
-void ff_vvc_sao_edge_filter_8x8_8_neon(uint8_t *dst, const uint8_t *src, ptrdiff_t stride_dst,
- const int16_t *sao_offset_val, int eo, int width, int height);
+#define SAO_EDGE_FILTER_PROTO(codec, size, depth) \
+ void ff_##codec##_sao_edge_filter_##size##_##depth##_neon( \
+ uint8_t *dst, const uint8_t *src, ptrdiff_t stride_dst, \
+ const int16_t *sao_offset_val, int eo, int width, int height)
+
+SAO_EDGE_FILTER_PROTO(hevc, 16x16, 8);
+SAO_EDGE_FILTER_PROTO(hevc, 8x8, 8);
+SAO_EDGE_FILTER_PROTO(hevc, 16x16, 10);
+SAO_EDGE_FILTER_PROTO(hevc, 8x8, 10);
+SAO_EDGE_FILTER_PROTO(hevc, 16x16, 12);
+SAO_EDGE_FILTER_PROTO(hevc, 8x8, 12);
+SAO_EDGE_FILTER_PROTO(vvc, 16x16, 8);
+SAO_EDGE_FILTER_PROTO(vvc, 8x8, 8);
+
+#undef SAO_EDGE_FILTER_PROTO
#define NEON8_FNPROTO_PARTIAL_6(fn, args, ext) \
void ff_hevc_put_hevc_##fn##_h4_8_neon##ext args; \
diff --git a/libavcodec/aarch64/h26x/sao_neon.S b/libavcodec/aarch64/h26x/sao_neon.S
index 90022fcfc7..d6a296db17 100644
--- a/libavcodec/aarch64/h26x/sao_neon.S
+++ b/libavcodec/aarch64/h26x/sao_neon.S
@@ -100,12 +100,18 @@ function ff_h26x_sao_band_filter_16x16_8_neon, export=1
ret
endfunc
-.Lhevc_sao_edge_pos:
+.Lhevc_sao_edge_pos_8:
.word 1 // horizontal
.word HEVC_SAO_STRIDE // vertical
.word HEVC_SAO_STRIDE + 1 // 135 degree
.word HEVC_SAO_STRIDE - 1 // 45 degree
+.Lhevc_sao_edge_pos_16bit:
+.word 2 // horizontal
+.word HEVC_SAO_STRIDE // vertical
+.word HEVC_SAO_STRIDE + 2 // 135 degree
+.word HEVC_SAO_STRIDE - 2 // 45 degree
+
.Lvvc_sao_edge_pos:
.word 1 // horizontal
.word VVC_SAO_STRIDE // vertical
@@ -121,7 +127,7 @@ endfunc
// ff_hevc_sao_edge_filter_16x16_8_neon(char *dst, char *src, ptrdiff stride_dst,
// int16 *sao_offset_val, int eo, int width, int height)
function ff_hevc_sao_edge_filter_16x16_8_neon, export=1
- adr x7, .Lhevc_sao_edge_pos
+ adr x7, .Lhevc_sao_edge_pos_8
mov x15, #HEVC_SAO_STRIDE
1:
ld1 {v3.8h}, [x3] // load sao_offset_val
@@ -179,6 +185,84 @@ function ff_hevc_sao_edge_filter_16x16_8_neon, export=1
ret
endfunc
+/* Permutes the low byte of sao_offset_val into the byte table v28 by
+ * edge_idx = { 1, 2, 0, 3, 4 }.
+ * v29: bias 2
+ * v30: 0 for clip lower bound
+ * v31: clip higher bound, not set by this macro
+ */
+.macro sao_edge_offsets_init
+ ld1 {v6.8h}, [x3]
+ mov x11, #0x0402
+ movi v29.8b, #2
+ movk x11, #0x0600, lsl #16
+ movi v30.8h, #0
+ movk x11, #0x0008, lsl #32 // 0x0000000806000402
+ fmov d7, x11
+ tbl v28.8b, {v6.16b}, v7.8b
+.endm
+
+/* Filters one 8-pixel chunk
+ * v0: cur, and hold clipped result
+ * v1: a
+ * v2: b
+ */
+.macro sao_edge_filter8
+ cmhi v16.8h, v1.8h, v0.8h
+ cmhi v17.8h, v0.8h, v1.8h
+ cmhi v18.8h, v2.8h, v0.8h
+ cmhi v19.8h, v0.8h, v2.8h
+ add v20.8h, v16.8h, v18.8h
+ add v21.8h, v17.8h, v19.8h
+ sub v20.8h, v20.8h, v21.8h // CMP(cur, a) + CMP(cur, b)
+ xtn v20.8b, v20.8h
+ add v20.8b, v20.8b, v29.8b // offset table index
+ tbl v16.8b, {v28.16b}, v20.8b
+ saddw v0.8h, v0.8h, v16.8b
+ smax v0.8h, v0.8h, v30.8h
+ smin v0.8h, v0.8h, v31.8h
+.endm
+
+// ff_hevc_sao_edge_filter_16x16_12_neon(char *dst, char *src, ptrdiff stride_dst,
+// int16 *sao_offset_val, int eo, int width, int height)
+function ff_hevc_sao_edge_filter_16x16_12_neon, export=1
+ mvni v31.8h, #0xf0, lsl #8 // 4095
+ b .Lhevc_sao_edge_16bit
+endfunc
+
+// ff_hevc_sao_edge_filter_16x16_10_neon(char *dst, char *src, ptrdiff stride_dst,
+// int16 *sao_offset_val, int eo, int width, int height)
+function ff_hevc_sao_edge_filter_16x16_10_neon, export=1
+ mvni v31.8h, #0xfc, lsl #8 // 1023
+.Lhevc_sao_edge_16bit:
+ adr x7, .Lhevc_sao_edge_pos_16bit
+ mov x15, #HEVC_SAO_STRIDE
+ add w5, w5, #7
+ ldr w4, [x7, w4, uxtw #2] // a/b offsets in bytes
+ bic w5, w5, #7
+ sao_edge_offsets_init
+ lsl w5, w5, #1 // width in bytes, multiple of 16
+ sub x15, x15, x5 // src step to the next line
+ sub x16, x2, x5 // dst step to the next line
+1:
+ lsr x14, x5, #4 // 8-pixel groups
+ sub x12, x1, x4
+ add x13, x1, x4
+2:
+ ldr q0, [x1], #16
+ ldr q1, [x12], #16
+ ldr q2, [x13], #16
+ subs x14, x14, #1
+ sao_edge_filter8
+ str q0, [x0], #16
+ b.ne 2b
+ subs w6, w6, #1
+ add x1, x1, x15
+ add x0, x0, x16
+ b.ne 1b
+ ret
+endfunc
+
function ff_vvc_sao_edge_filter_8x8_8_neon, export=1
adr x7, .Lvvc_sao_edge_pos
mov x15, #VVC_SAO_STRIDE
@@ -188,7 +272,7 @@ endfunc
// ff_hevc_sao_edge_filter_8x8_8_neon(char *dst, char *src, ptrdiff stride_dst,
// int16 *sao_offset_val, int eo, int width, int height)
function ff_hevc_sao_edge_filter_8x8_8_neon, export=1
- adr x7, .Lhevc_sao_edge_pos
+ adr x7, .Lhevc_sao_edge_pos_8
mov x15, #HEVC_SAO_STRIDE
1:
ldr w4, [x7, w4, uxtw #2]
@@ -235,3 +319,32 @@ function ff_hevc_sao_edge_filter_8x8_8_neon, export=1
b.ne 1b
ret
endfunc
+
+// ff_hevc_sao_edge_filter_8x8_12_neon(char *dst, char *src, ptrdiff stride_dst,
+// int16 *sao_offset_val, int eo, int width, int height)
+function ff_hevc_sao_edge_filter_8x8_12_neon, export=1
+ mvni v31.8h, #0xf0, lsl #8 // 4095
+ b .Lhevc_sao_edge_8x8_16bit
+endfunc
+
+// ff_hevc_sao_edge_filter_8x8_10_neon(char *dst, char *src, ptrdiff stride_dst,
+// int16 *sao_offset_val, int eo, int width, int height)
+function ff_hevc_sao_edge_filter_8x8_10_neon, export=1
+ mvni v31.8h, #0xfc, lsl #8 // 1023
+.Lhevc_sao_edge_8x8_16bit:
+ adr x7, .Lhevc_sao_edge_pos_16bit
+ mov x15, #HEVC_SAO_STRIDE
+ ldr w4, [x7, w4, uxtw #2] // a/b offsets in bytes
+ sao_edge_offsets_init
+ sub x9, x1, x4 // a neighbours
+ add x10, x1, x4 // b neighbours
+1:
+ ld1 {v0.8h}, [x1], x15
+ ld1 {v1.8h}, [x9], x15
+ ld1 {v2.8h}, [x10], x15
+ subs w6, w6, #1
+ sao_edge_filter8
+ st1 {v0.8h}, [x0], x2
+ b.ne 1b
+ ret
+endfunc
diff --git a/libavcodec/aarch64/hevcdsp_init_aarch64.c b/libavcodec/aarch64/hevcdsp_init_aarch64.c
index a2ca8aa124..12f8d93803 100644
--- a/libavcodec/aarch64/hevcdsp_init_aarch64.c
+++ b/libavcodec/aarch64/hevcdsp_init_aarch64.c
@@ -338,6 +338,11 @@ av_cold void ff_hevc_dsp_init_aarch64(HEVCDSPContext *c, const int bit_depth)
c->idct_dc[2] = ff_hevc_idct_16x16_dc_10_neon;
c->idct_dc[3] = ff_hevc_idct_32x32_dc_10_neon;
c->dequant = hevc_dequant_10_neon;
+ c->sao_edge_filter[0] = ff_hevc_sao_edge_filter_8x8_10_neon;
+ c->sao_edge_filter[1] =
+ c->sao_edge_filter[2] =
+ c->sao_edge_filter[3] =
+ c->sao_edge_filter[4] = ff_hevc_sao_edge_filter_16x16_10_neon;
}
if (bit_depth == 12) {
c->hevc_h_loop_filter_luma = ff_hevc_h_loop_filter_luma_12_neon;
@@ -353,5 +358,10 @@ av_cold void ff_hevc_dsp_init_aarch64(HEVCDSPContext *c, const int bit_depth)
c->idct_dc[2] = ff_hevc_idct_16x16_dc_12_neon;
c->idct_dc[3] = ff_hevc_idct_32x32_dc_12_neon;
c->dequant = hevc_dequant_12_neon;
+ c->sao_edge_filter[0] = ff_hevc_sao_edge_filter_8x8_12_neon;
+ c->sao_edge_filter[1] =
+ c->sao_edge_filter[2] =
+ c->sao_edge_filter[3] =
+ c->sao_edge_filter[4] = ff_hevc_sao_edge_filter_16x16_12_neon;
}
}