Commit fc3ff7734f for ffmpeg
commit fc3ff7734f7ea668fa3da5ec5002e609a6bd44d2
Author: Charly Morgand-Poyac <charly.morgand-poyac@bbright.com>
Date: Mon Jul 6 16:31:07 2026 +0200
lavc/hevcdec: decode tiles in parallel under slice threading
Each tile of a whole-frame single-slice picture is an independent
substream; decode them concurrently, then filter the frame serially.
Around 2.3x faster (up to 3.5x) at 8 threads on tiled streams.
Signed-off-by: Charly Morgand-Poyac <charly.morgand-poyac@bbright.com>
diff --git a/libavcodec/hevc/filter.c b/libavcodec/hevc/filter.c
index e897ad5d58..c8e9fa8bd6 100644
--- a/libavcodec/hevc/filter.c
+++ b/libavcodec/hevc/filter.c
@@ -760,7 +760,7 @@ void ff_hevc_deblocking_boundary_strengths(HEVCLocalContext *lc, const HEVCLayer
((!s->sh.slice_loop_filter_across_slices_enabled_flag &&
lc->boundary_flags & BOUNDARY_UPPER_SLICE &&
(y0 % (1 << sps->log2_ctb_size)) == 0) ||
- (!pps->loop_filter_across_tiles_enabled_flag &&
+ ((!pps->loop_filter_across_tiles_enabled_flag || lc->tile_bs_defer) &&
lc->boundary_flags & BOUNDARY_UPPER_TILE &&
(y0 % (1 << sps->log2_ctb_size)) == 0)))
boundary_upper = 0;
@@ -798,7 +798,7 @@ void ff_hevc_deblocking_boundary_strengths(HEVCLocalContext *lc, const HEVCLayer
((!s->sh.slice_loop_filter_across_slices_enabled_flag &&
lc->boundary_flags & BOUNDARY_LEFT_SLICE &&
(x0 % (1 << sps->log2_ctb_size)) == 0) ||
- (!pps->loop_filter_across_tiles_enabled_flag &&
+ ((!pps->loop_filter_across_tiles_enabled_flag || lc->tile_bs_defer) &&
lc->boundary_flags & BOUNDARY_LEFT_TILE &&
(x0 % (1 << sps->log2_ctb_size)) == 0)))
boundary_left = 0;
@@ -865,6 +865,68 @@ void ff_hevc_deblocking_boundary_strengths(HEVCLocalContext *lc, const HEVCLayer
}
}
+void ff_hevc_tile_boundary_bs(HEVCLocalContext *lc, const HEVCLayerContext *l,
+ const HEVCPPS *pps, int x0, int y0)
+{
+ const HEVCSPS *const sps = pps->sps;
+ const HEVCContext *s = lc->parent;
+ const MvField *tab_mvf = s->cur_frame->tab_mvf;
+ const int log2_min_pu_size = sps->log2_min_pu_size;
+ const int log2_min_tu_size = sps->log2_min_tb_size;
+ const int min_pu_width = sps->min_pu_width;
+ const int min_tu_width = sps->min_tb_width;
+ const int ctb_size = 1 << sps->log2_ctb_size;
+ int i, bs;
+
+ if ((lc->boundary_flags & BOUNDARY_UPPER_TILE) && y0 > 0) {
+ const RefPicList *rpl_top = (lc->boundary_flags & BOUNDARY_UPPER_SLICE) ?
+ ff_hevc_get_ref_list(s->cur_frame, x0, y0 - 1) :
+ s->cur_frame->refPicList;
+ int yp_pu = (y0 - 1) >> log2_min_pu_size, yq_pu = y0 >> log2_min_pu_size;
+ int yp_tu = (y0 - 1) >> log2_min_tu_size, yq_tu = y0 >> log2_min_tu_size;
+ int len = FFMIN(ctb_size, sps->width - x0);
+ for (i = 0; i < len; i += 4) {
+ int x_pu = (x0 + i) >> log2_min_pu_size;
+ int x_tu = (x0 + i) >> log2_min_tu_size;
+ const MvField *top = &tab_mvf[yp_pu * min_pu_width + x_pu];
+ const MvField *curr = &tab_mvf[yq_pu * min_pu_width + x_pu];
+ uint8_t top_cbf = l->cbf_luma[yp_tu * min_tu_width + x_tu];
+ uint8_t curr_cbf = l->cbf_luma[yq_tu * min_tu_width + x_tu];
+ if (curr->pred_flag == PF_INTRA || top->pred_flag == PF_INTRA)
+ bs = 2;
+ else if (curr_cbf || top_cbf)
+ bs = 1;
+ else
+ bs = boundary_strength(s, curr, top, rpl_top);
+ l->horizontal_bs[((x0 + i) + y0 * l->bs_width) >> 2] = bs;
+ }
+ }
+
+ if ((lc->boundary_flags & BOUNDARY_LEFT_TILE) && x0 > 0) {
+ const RefPicList *rpl_left = (lc->boundary_flags & BOUNDARY_LEFT_SLICE) ?
+ ff_hevc_get_ref_list(s->cur_frame, x0 - 1, y0) :
+ s->cur_frame->refPicList;
+ int xp_pu = (x0 - 1) >> log2_min_pu_size, xq_pu = x0 >> log2_min_pu_size;
+ int xp_tu = (x0 - 1) >> log2_min_tu_size, xq_tu = x0 >> log2_min_tu_size;
+ int len = FFMIN(ctb_size, sps->height - y0);
+ for (i = 0; i < len; i += 4) {
+ int y_pu = (y0 + i) >> log2_min_pu_size;
+ int y_tu = (y0 + i) >> log2_min_tu_size;
+ const MvField *left = &tab_mvf[y_pu * min_pu_width + xp_pu];
+ const MvField *curr = &tab_mvf[y_pu * min_pu_width + xq_pu];
+ uint8_t left_cbf = l->cbf_luma[y_tu * min_tu_width + xp_tu];
+ uint8_t curr_cbf = l->cbf_luma[y_tu * min_tu_width + xq_tu];
+ if (curr->pred_flag == PF_INTRA || left->pred_flag == PF_INTRA)
+ bs = 2;
+ else if (curr_cbf || left_cbf)
+ bs = 1;
+ else
+ bs = boundary_strength(s, curr, left, rpl_left);
+ l->vertical_bs[(x0 + (y0 + i) * l->bs_width) >> 2] = bs;
+ }
+ }
+}
+
#undef LUMA
#undef CB
#undef CR
diff --git a/libavcodec/hevc/hevcdec.c b/libavcodec/hevc/hevcdec.c
index 2ab6b3e2c0..b572d8dd11 100644
--- a/libavcodec/hevc/hevcdec.c
+++ b/libavcodec/hevc/hevcdec.c
@@ -2733,7 +2733,9 @@ static void hls_decode_neighbour(HEVCLocalContext *lc,
int ctb_addr_rs = pps->ctb_addr_ts_to_rs[ctb_addr_ts];
int ctb_addr_in_slice = ctb_addr_rs - s->sh.slice_addr;
- l->tab_slice_address[ctb_addr_rs] = s->sh.slice_addr;
+ /* the tile-parallel path pre-fills this serially, workers only read it */
+ if (!lc->tile_bs_defer)
+ l->tab_slice_address[ctb_addr_rs] = s->sh.slice_addr;
if (pps->entropy_coding_sync_enabled_flag) {
if (x_ctb == 0 && (y_ctb & (ctb_size - 1)) == 0)
@@ -3070,6 +3072,121 @@ static int hls_slice_data_wpp(HEVCContext *s, const H2645NAL *nal)
return res;
}
+static int hls_decode_entry_tile(AVCodecContext *avctx, void *hevc_lclist,
+ int job, int thread)
+{
+ HEVCLocalContext *lc = &((HEVCLocalContext*)hevc_lclist)[thread];
+ const HEVCContext *const s = lc->parent;
+ const HEVCLayerContext *const l = &s->layers[s->cur_layer];
+ const HEVCPPS *const pps = s->pps;
+ const HEVCSPS *const sps = pps->sps;
+ const uint8_t *data = s->data + s->sh.offset[job];
+ const size_t data_size = s->sh.size[job];
+ /* the slice covers every tile, so job and tile are the same index */
+ int ctb_addr_ts = pps->ctb_addr_rs_to_ts[pps->row_bd[job / pps->num_tile_columns] * sps->ctb_width +
+ pps->col_bd[job % pps->num_tile_columns]];
+ int more_data = 1, ret;
+
+ lc->tile_bs_defer = 1;
+ lc->tu.cu_qp_offset_cb = 0;
+ lc->tu.cu_qp_offset_cr = 0;
+ /* hls_decode_neighbour() skips this for the first CTB of the picture */
+ lc->end_of_tiles_x = pps->col_bd[job % pps->num_tile_columns + 1] << sps->log2_ctb_size;
+
+ while (more_data && ctb_addr_ts < sps->ctb_size &&
+ pps->tile_id[ctb_addr_ts] == job) {
+ int ctb_addr_rs = pps->ctb_addr_ts_to_rs[ctb_addr_ts];
+ int x_ctb = (ctb_addr_rs % sps->ctb_width) << sps->log2_ctb_size;
+ int y_ctb = (ctb_addr_rs / sps->ctb_width) << sps->log2_ctb_size;
+
+ hls_decode_neighbour(lc, l, pps, sps, x_ctb, y_ctb, ctb_addr_ts);
+
+ ret = ff_hevc_cabac_init(lc, pps, ctb_addr_ts, data, data_size, 1);
+ if (ret < 0)
+ return ret;
+
+ hls_sao_param(lc, l, pps, sps,
+ x_ctb >> sps->log2_ctb_size, y_ctb >> sps->log2_ctb_size);
+
+ l->deblock[ctb_addr_rs].beta_offset = s->sh.beta_offset;
+ l->deblock[ctb_addr_rs].tc_offset = s->sh.tc_offset;
+ l->filter_slice_edges[ctb_addr_rs] = s->sh.slice_loop_filter_across_slices_enabled_flag;
+
+ more_data = hls_coding_quadtree(lc, l, pps, sps, x_ctb, y_ctb, sps->log2_ctb_size, 0);
+ if (more_data < 0)
+ return more_data;
+ ctb_addr_ts++;
+ }
+ return ctb_addr_ts;
+}
+
+static int hls_slice_data_tiles(HEVCContext *s, const H2645NAL *nal)
+{
+ const HEVCPPS *const pps = s->pps;
+ const HEVCSPS *const sps = pps->sps;
+ const HEVCLayerContext *const l = &s->layers[s->cur_layer];
+ const int ctb_size = 1 << sps->log2_ctb_size;
+ const int nb_tiles = s->sh.num_entry_point_offsets + 1;
+ const int start_ts = pps->ctb_addr_rs_to_ts[s->sh.slice_ctb_addr_rs];
+ int res = 0, ctb_addr_ts, x_ctb = 0, y_ctb = 0;
+ int *ret;
+
+ res = alloc_local_ctxs(s);
+ if (res < 0)
+ return res;
+
+ res = slice_substreams_init(s, nal);
+ if (res < 0)
+ return res;
+
+ for (unsigned i = 1; i < s->nb_local_ctx; i++) {
+ s->local_ctx[i].first_qp_group = 1;
+ s->local_ctx[i].qp_y = s->local_ctx[0].qp_y;
+ }
+
+ for (ctb_addr_ts = start_ts; ctb_addr_ts < sps->ctb_size; ctb_addr_ts++)
+ l->tab_slice_address[pps->ctb_addr_ts_to_rs[ctb_addr_ts]] = s->sh.slice_addr;
+
+ ret = av_calloc(nb_tiles, sizeof(*ret));
+ if (!ret)
+ return AVERROR(ENOMEM);
+ s->avctx->execute2(s->avctx, hls_decode_entry_tile, s->local_ctx, ret, nb_tiles);
+ for (int i = 0; i < nb_tiles; i++)
+ if (ret[i] < 0)
+ res = ret[i];
+ av_free(ret);
+
+ for (unsigned i = 0; i < s->nb_local_ctx; i++)
+ s->local_ctx[i].tile_bs_defer = 0;
+
+ if (res < 0)
+ return res;
+
+ if (pps->loop_filter_across_tiles_enabled_flag &&
+ !s->sh.disable_deblocking_filter_flag) {
+ for (ctb_addr_ts = start_ts; ctb_addr_ts < sps->ctb_size; ctb_addr_ts++) {
+ int ctb_addr_rs = pps->ctb_addr_ts_to_rs[ctb_addr_ts];
+ x_ctb = (ctb_addr_rs % sps->ctb_width) << sps->log2_ctb_size;
+ y_ctb = (ctb_addr_rs / sps->ctb_width) << sps->log2_ctb_size;
+ hls_decode_neighbour(&s->local_ctx[0], l, pps, sps, x_ctb, y_ctb, ctb_addr_ts);
+ if (s->local_ctx[0].boundary_flags & (BOUNDARY_LEFT_TILE | BOUNDARY_UPPER_TILE))
+ ff_hevc_tile_boundary_bs(&s->local_ctx[0], l, pps, x_ctb, y_ctb);
+ }
+ }
+
+ for (ctb_addr_ts = start_ts; ctb_addr_ts < sps->ctb_size; ctb_addr_ts++) {
+ int ctb_addr_rs = pps->ctb_addr_ts_to_rs[ctb_addr_ts];
+ x_ctb = (ctb_addr_rs % sps->ctb_width) << sps->log2_ctb_size;
+ y_ctb = (ctb_addr_rs / sps->ctb_width) << sps->log2_ctb_size;
+ hls_decode_neighbour(&s->local_ctx[0], l, pps, sps, x_ctb, y_ctb, ctb_addr_ts);
+ ff_hevc_hls_filters(&s->local_ctx[0], l, pps, x_ctb, y_ctb, ctb_size);
+ }
+ if (x_ctb + ctb_size >= sps->width && y_ctb + ctb_size >= sps->height)
+ ff_hevc_hls_filter(&s->local_ctx[0], l, pps, x_ctb, y_ctb, ctb_size);
+
+ return sps->ctb_size;
+}
+
static int decode_slice_data(HEVCContext *s, const HEVCLayerContext *l,
const H2645NAL *nal, GetBitContext *gb)
{
@@ -3124,6 +3241,14 @@ static int decode_slice_data(HEVCContext *s, const HEVCLayerContext *l,
pps->num_tile_rows == 1 && pps->num_tile_columns == 1)
return hls_slice_data_wpp(s, nal);
+ if (s->avctx->active_thread_type == FF_THREAD_SLICE &&
+ s->sh.num_entry_point_offsets > 0 &&
+ pps->tiles_enabled_flag &&
+ !pps->entropy_coding_sync_enabled_flag &&
+ s->sh.first_slice_in_pic_flag &&
+ s->sh.num_entry_point_offsets + 1 == pps->num_tile_rows * pps->num_tile_columns)
+ return hls_slice_data_tiles(s, nal);
+
return hls_decode_entry(s, gb);
}
diff --git a/libavcodec/hevc/hevcdec.h b/libavcodec/hevc/hevcdec.h
index 9e5acbd767..703b8df1af 100644
--- a/libavcodec/hevc/hevcdec.h
+++ b/libavcodec/hevc/hevcdec.h
@@ -444,6 +444,9 @@ typedef struct HEVCLocalContext {
* of the deblocking filter */
int boundary_flags;
+ /* decoding tiles in parallel: defer tile-boundary BS to a serial pass */
+ int tile_bs_defer;
+
// an array of these structs is used for per-thread state - pad its size
// to avoid false sharing
char padding[128];
@@ -710,6 +713,8 @@ void ff_hevc_set_qPy(HEVCLocalContext *lc,
void ff_hevc_deblocking_boundary_strengths(HEVCLocalContext *lc, const HEVCLayerContext *l,
const HEVCPPS *pps,
int x0, int y0, int log2_trafo_size);
+void ff_hevc_tile_boundary_bs(HEVCLocalContext *lc, const HEVCLayerContext *l,
+ const HEVCPPS *pps, int x0, int y0);
int ff_hevc_cu_qp_delta_sign_flag(HEVCLocalContext *lc);
int ff_hevc_cu_qp_delta_abs(HEVCLocalContext *lc);
int ff_hevc_cu_chroma_qp_offset_flag(HEVCLocalContext *lc);
diff --git a/tests/fate/hevc.mak b/tests/fate/hevc.mak
index ba5242e0d2..b39358167e 100644
--- a/tests/fate/hevc.mak
+++ b/tests/fate/hevc.mak
@@ -303,6 +303,15 @@ FATE_HEVC_FFPROBE-$(call DEMDEC, HEVC, HEVC) += fate-hevc-skip-pred-fields
fate-hevc-skip-pred-pts: CMD = probeframes -show_entries frame=key_frame,pts,pict_type -skip_pred all -skip_idct all $(TARGET_SAMPLES)/mov/elst_ends_betn_b_and_i.mp4
FATE_HEVC_FFPROBE-$(call DEMDEC, MOV, HEVC) += fate-hevc-skip-pred-pts
+# TILES_A conformance picture through the tile-parallel slice path, same output
+fate-hevc-tiles-slice-threads: CMD = thread_type=slice threads=4 framecrc -i $(TARGET_SAMPLES)/hevc-conformance/TILES_A_Cisco_2.bit -pix_fmt yuv420p
+fate-hevc-tiles-slice-threads: REF = $(SRC_PATH)/tests/ref/fate/hevc-conformance-TILES_A_Cisco_2
+FATE_HEVC-$(call FRAMECRC, HEVC, HEVC, HEVC_PARSER) += fate-hevc-tiles-slice-threads
+
+# deblocking disabled + loop_filter_across_tiles: tile edges must stay unfiltered
+fate-hevc-tiles-nodeblock-slice-threads: CMD = thread_type=slice threads=4 framecrc -i $(TARGET_SAMPLES)/hevc/tiles_nodeblock.hevc
+FATE_HEVC-$(call FRAMECRC, HEVC, HEVC, HEVC_PARSER) += fate-hevc-tiles-nodeblock-slice-threads
+
fate-hevc-cabac-tudepth: CMD = framecrc -i $(TARGET_SAMPLES)/hevc/cbf_cr_cb_TUDepth_4_circle.h265 -pix_fmt yuv444p
FATE_HEVC-$(call FRAMECRC, HEVC, HEVC) += fate-hevc-cabac-tudepth
diff --git a/tests/ref/fate/hevc-tiles-nodeblock-slice-threads b/tests/ref/fate/hevc-tiles-nodeblock-slice-threads
new file mode 100644
index 0000000000..41cd7f1862
--- /dev/null
+++ b/tests/ref/fate/hevc-tiles-nodeblock-slice-threads
@@ -0,0 +1,20 @@
+#tb 0: 1/30
+#media_type 0: video
+#codec_id 0: rawvideo
+#dimensions 0: 832x480
+#sar 0: 0/1
+0, 0, 0, 1, 599040, 0x0c2d0ff4
+0, 1, 1, 1, 599040, 0xbb240afc
+0, 2, 2, 1, 599040, 0x2a587dd1
+0, 3, 3, 1, 599040, 0x63d9f737
+0, 4, 4, 1, 599040, 0xe62185e1
+0, 5, 5, 1, 599040, 0x3640fd36
+0, 6, 6, 1, 599040, 0x71a7f607
+0, 7, 7, 1, 599040, 0xf7a797ae
+0, 8, 8, 1, 599040, 0x743eddfb
+0, 9, 9, 1, 599040, 0x2e1d612d
+0, 10, 10, 1, 599040, 0x97d2b583
+0, 11, 11, 1, 599040, 0x6186fbcc
+0, 12, 12, 1, 599040, 0xd963b62f
+0, 13, 13, 1, 599040, 0x47b22c72
+0, 14, 14, 1, 599040, 0x6a73ff19