Commit 26c6816846 for ffmpeg
commit 26c68168461d57519b18fbceac92a58fe7ad1db6
Author: Zhao Zhili <quinkblack@foxmail.com>
Date: Tue Sep 15 11:46:01 2026 +0800
avcodec/aarch64: add NEON VVC SAO edge 10 and 12 bit
Cortex-A510 Cortex-A715 Cortex-A725 Cortex-X3 Cortex-X925
vvc_sao_edge_8_10_neon: 308.9 (4.29x) 57.1 (5.47x) 57.4 (4.66x) 36.4 (7.35x) 23.7 (8.13x)
vvc_sao_edge_8_12_neon: 314.6 (4.19x) 57.3 (5.58x) 58.3 (4.59x) 36.4 (7.34x) 23.9 (8.08x)
vvc_sao_edge_16_10_neon: 1202.1 (4.28x) 228.1 (5.26x) 228.6 (4.54x) 144.6 (7.28x) 95.0 (8.39x)
vvc_sao_edge_16_12_neon: 1203.3 (4.28x) 229.4 (5.37x) 239.4 (4.33x) 144.6 (7.22x) 95.0 (8.38x)
vvc_sao_edge_32_10_neon: 4718.8 (4.28x) 913.1 (5.12x) 911.5 (4.42x) 571.5 (7.30x) 366.9 (9.08x)
vvc_sao_edge_32_12_neon: 4737.9 (4.26x) 913.3 (5.28x) 913.0 (4.41x) 571.8 (7.31x) 365.6 (9.15x)
vvc_sao_edge_48_10_neon: 10970.3 (4.16x) 2035.0 (5.13x) 2036.2 (4.43x) 1307.6 (7.14x) 823.6 (9.19x)
vvc_sao_edge_48_12_neon: 10980.7 (4.15x) 2037.3 (5.29x) 2035.0 (4.43x) 1311.2 (7.14x) 824.1 (9.20x)
vvc_sao_edge_64_10_neon: 19127.9 (4.23x) 3605.3 (5.13x) 3628.5 (4.41x) 2315.9 (7.17x) 1452.7 (9.33x)
vvc_sao_edge_64_12_neon: 19136.2 (4.23x) 3606.3 (5.29x) 3629.0 (4.40x) 2320.9 (7.17x) 1448.9 (9.36x)
vvc_sao_edge_80_10_neon: 29560.9 (4.29x) 5624.0 (5.30x) 5657.5 (4.64x) 3610.8 (7.17x) 2252.1 (9.44x)
vvc_sao_edge_80_12_neon: 29644.8 (4.28x) 5626.8 (5.48x) 5661.2 (4.64x) 3614.4 (7.18x) 2250.3 (9.46x)
vvc_sao_edge_96_10_neon: 42316.0 (4.29x) 8088.8 (5.28x) 8135.3 (4.61x) 5188.5 (7.20x) 3236.3 (9.49x)
vvc_sao_edge_96_12_neon: 42863.0 (4.23x) 8091.9 (5.45x) 8136.9 (4.61x) 5190.6 (7.21x) 3236.1 (9.48x)
vvc_sao_edge_112_10_neon: 57410.2 (4.30x) 11001.8 (5.27x) 11056.2 (4.59x) 7066.3 (7.19x) 4407.8 (9.48x)
vvc_sao_edge_112_12_neon: 57372.5 (4.30x) 11006.7 (5.44x) 11054.2 (4.59x) 7074.3 (7.19x) 4400.6 (9.52x)
vvc_sao_edge_128_10_neon: 75737.0 (4.26x) 14401.4 (5.26x) 14468.5 (4.57x) 9227.8 (7.22x) 5729.6 (9.54x)
vvc_sao_edge_128_12_neon: 75824.1 (4.25x) 14404.0 (5.42x) 14472.1 (4.57x) 9222.1 (7.20x) 5730.8 (9.55x)
Signed-off-by: Zhao Zhili <zhilizhao@tencent.com>
diff --git a/libavcodec/aarch64/h26x/dsp.h b/libavcodec/aarch64/h26x/dsp.h
index dab5163c60..2b8eae130a 100644
--- a/libavcodec/aarch64/h26x/dsp.h
+++ b/libavcodec/aarch64/h26x/dsp.h
@@ -45,6 +45,10 @@ SAO_EDGE_FILTER_PROTO(hevc, 16x16, 12);
SAO_EDGE_FILTER_PROTO(hevc, 8x8, 12);
SAO_EDGE_FILTER_PROTO(vvc, 16x16, 8);
SAO_EDGE_FILTER_PROTO(vvc, 8x8, 8);
+SAO_EDGE_FILTER_PROTO(vvc, 16x16, 10);
+SAO_EDGE_FILTER_PROTO(vvc, 8x8, 10);
+SAO_EDGE_FILTER_PROTO(vvc, 16x16, 12);
+SAO_EDGE_FILTER_PROTO(vvc, 8x8, 12);
#undef SAO_EDGE_FILTER_PROTO
diff --git a/libavcodec/aarch64/h26x/sao_neon.S b/libavcodec/aarch64/h26x/sao_neon.S
index d6a296db17..083f0eb89d 100644
--- a/libavcodec/aarch64/h26x/sao_neon.S
+++ b/libavcodec/aarch64/h26x/sao_neon.S
@@ -118,6 +118,12 @@ endfunc
.word VVC_SAO_STRIDE + 1 // 135 degree
.word VVC_SAO_STRIDE - 1 // 45 degree
+.Lvvc_sao_edge_pos_16bit:
+.word 2 // horizontal
+.word VVC_SAO_STRIDE // vertical
+.word VVC_SAO_STRIDE + 2 // 135 degree
+.word VVC_SAO_STRIDE - 2 // 45 degree
+
function ff_vvc_sao_edge_filter_16x16_8_neon, export=1
adr x7, .Lvvc_sao_edge_pos
mov x15, #VVC_SAO_STRIDE
@@ -223,6 +229,23 @@ endfunc
smin v0.8h, v0.8h, v31.8h
.endm
+// ff_vvc_sao_edge_filter_16x16_12_neon(char *dst, char *src, ptrdiff stride_dst,
+// int16 *sao_offset_val, int eo, int width, int height)
+function ff_vvc_sao_edge_filter_16x16_12_neon, export=1
+ mvni v31.8h, #0xf0, lsl #8 // 4095
+ b .Lvvc_sao_edge_16x16_16bit
+endfunc
+
+// ff_vvc_sao_edge_filter_16x16_10_neon(char *dst, char *src, ptrdiff stride_dst,
+// int16 *sao_offset_val, int eo, int width, int height)
+function ff_vvc_sao_edge_filter_16x16_10_neon, export=1
+ mvni v31.8h, #0xfc, lsl #8 // 1023
+.Lvvc_sao_edge_16x16_16bit:
+ adr x7, .Lvvc_sao_edge_pos_16bit
+ mov x15, #VVC_SAO_STRIDE
+ b .Lh26x_sao_edge_16x16_16bit
+endfunc
+
// ff_hevc_sao_edge_filter_16x16_12_neon(char *dst, char *src, ptrdiff stride_dst,
// int16 *sao_offset_val, int eo, int width, int height)
function ff_hevc_sao_edge_filter_16x16_12_neon, export=1
@@ -237,6 +260,7 @@ function ff_hevc_sao_edge_filter_16x16_10_neon, export=1
.Lhevc_sao_edge_16bit:
adr x7, .Lhevc_sao_edge_pos_16bit
mov x15, #HEVC_SAO_STRIDE
+.Lh26x_sao_edge_16x16_16bit:
add w5, w5, #7
ldr w4, [x7, w4, uxtw #2] // a/b offsets in bytes
bic w5, w5, #7
@@ -320,6 +344,23 @@ function ff_hevc_sao_edge_filter_8x8_8_neon, export=1
ret
endfunc
+// ff_vvc_sao_edge_filter_8x8_12_neon(char *dst, char *src, ptrdiff stride_dst,
+// int16 *sao_offset_val, int eo, int width, int height)
+function ff_vvc_sao_edge_filter_8x8_12_neon, export=1
+ mvni v31.8h, #0xf0, lsl #8 // 4095
+ b .Lvvc_sao_edge_8x8_16bit
+endfunc
+
+// ff_vvc_sao_edge_filter_8x8_10_neon(char *dst, char *src, ptrdiff stride_dst,
+// int16 *sao_offset_val, int eo, int width, int height)
+function ff_vvc_sao_edge_filter_8x8_10_neon, export=1
+ mvni v31.8h, #0xfc, lsl #8 // 1023
+.Lvvc_sao_edge_8x8_16bit:
+ adr x7, .Lvvc_sao_edge_pos_16bit
+ mov x15, #VVC_SAO_STRIDE
+ b .Lh26x_sao_edge_8x8_16bit
+endfunc
+
// ff_hevc_sao_edge_filter_8x8_12_neon(char *dst, char *src, ptrdiff stride_dst,
// int16 *sao_offset_val, int eo, int width, int height)
function ff_hevc_sao_edge_filter_8x8_12_neon, export=1
@@ -334,6 +375,7 @@ function ff_hevc_sao_edge_filter_8x8_10_neon, export=1
.Lhevc_sao_edge_8x8_16bit:
adr x7, .Lhevc_sao_edge_pos_16bit
mov x15, #HEVC_SAO_STRIDE
+.Lh26x_sao_edge_8x8_16bit:
ldr w4, [x7, w4, uxtw #2] // a/b offsets in bytes
sao_edge_offsets_init
sub x9, x1, x4 // a neighbours
diff --git a/libavcodec/aarch64/vvc/dsp_init.c b/libavcodec/aarch64/vvc/dsp_init.c
index 53bf9f9edd..8936a4606a 100644
--- a/libavcodec/aarch64/vvc/dsp_init.c
+++ b/libavcodec/aarch64/vvc/dsp_init.c
@@ -372,6 +372,10 @@ void ff_vvc_dsp_init_aarch64(VVCDSPContext *const c, const int bd)
c->inter.put[1][5][1][1] =
c->inter.put[1][6][1][1] = ff_vvc_put_chroma_hv_x16_10_neon;
+ c->sao.edge_filter[0] = ff_vvc_sao_edge_filter_8x8_10_neon;
+ for (int i = 1; i < FF_ARRAY_ELEMS(c->sao.edge_filter); i++)
+ c->sao.edge_filter[i] = ff_vvc_sao_edge_filter_16x16_10_neon;
+
c->alf.filter[LUMA] = alf_filter_luma_10_neon;
c->alf.filter[CHROMA] = alf_filter_chroma_10_neon;
c->alf.classify = alf_classify_10_neon;
@@ -424,6 +428,10 @@ void ff_vvc_dsp_init_aarch64(VVCDSPContext *const c, const int bd)
c->inter.put[1][5][1][1] =
c->inter.put[1][6][1][1] = ff_vvc_put_chroma_hv_x16_12_neon;
+ c->sao.edge_filter[0] = ff_vvc_sao_edge_filter_8x8_12_neon;
+ for (int i = 1; i < FF_ARRAY_ELEMS(c->sao.edge_filter); i++)
+ c->sao.edge_filter[i] = ff_vvc_sao_edge_filter_16x16_12_neon;
+
c->alf.filter[LUMA] = alf_filter_luma_12_neon;
c->alf.filter[CHROMA] = alf_filter_chroma_12_neon;
c->alf.classify = alf_classify_12_neon;