This is an automated email from the git hooks/post-receive script. Git pushed a commit to branch master in repository ffmpeg.
commit fc3ff7734f7ea668fa3da5ec5002e609a6bd44d2 Author: Charly Morgand-Poyac <[email protected]> AuthorDate: Mon Jul 6 16:31:07 2026 +0200 Commit: michaelni <[email protected]> CommitDate: Sun Oct 4 20:45:29 2026 +0000 lavc/hevcdec: decode tiles in parallel under slice threading Each tile of a whole-frame single-slice picture is an independent substream; decode them concurrently, then filter the frame serially. Around 2.3x faster (up to 3.5x) at 8 threads on tiled streams. Signed-off-by: Charly Morgand-Poyac <[email protected]> --- libavcodec/hevc/filter.c | 66 ++++++++++- libavcodec/hevc/hevcdec.c | 127 +++++++++++++++++++++- libavcodec/hevc/hevcdec.h | 5 + tests/fate/hevc.mak | 9 ++ tests/ref/fate/hevc-tiles-nodeblock-slice-threads | 20 ++++ 5 files changed, 224 insertions(+), 3 deletions(-) diff --git a/libavcodec/hevc/filter.c b/libavcodec/hevc/filter.c index e897ad5d58..c8e9fa8bd6 100644 --- a/libavcodec/hevc/filter.c +++ b/libavcodec/hevc/filter.c @@ -760,7 +760,7 @@ void ff_hevc_deblocking_boundary_strengths(HEVCLocalContext *lc, const HEVCLayer ((!s->sh.slice_loop_filter_across_slices_enabled_flag && lc->boundary_flags & BOUNDARY_UPPER_SLICE && (y0 % (1 << sps->log2_ctb_size)) == 0) || - (!pps->loop_filter_across_tiles_enabled_flag && + ((!pps->loop_filter_across_tiles_enabled_flag || lc->tile_bs_defer) && lc->boundary_flags & BOUNDARY_UPPER_TILE && (y0 % (1 << sps->log2_ctb_size)) == 0))) boundary_upper = 0; @@ -798,7 +798,7 @@ void ff_hevc_deblocking_boundary_strengths(HEVCLocalContext *lc, const HEVCLayer ((!s->sh.slice_loop_filter_across_slices_enabled_flag && lc->boundary_flags & BOUNDARY_LEFT_SLICE && (x0 % (1 << sps->log2_ctb_size)) == 0) || - (!pps->loop_filter_across_tiles_enabled_flag && + ((!pps->loop_filter_across_tiles_enabled_flag || lc->tile_bs_defer) && lc->boundary_flags & BOUNDARY_LEFT_TILE && (x0 % (1 << sps->log2_ctb_size)) == 0))) boundary_left = 0; @@ -865,6 +865,68 @@ void ff_hevc_deblocking_boundary_strengths(HEVCLocalContext *lc, const HEVCLayer } } +void ff_hevc_tile_boundary_bs(HEVCLocalContext *lc, const HEVCLayerContext *l, + const HEVCPPS *pps, int x0, int y0) +{ + const HEVCSPS *const sps = pps->sps; + const HEVCContext *s = lc->parent; + const MvField *tab_mvf = s->cur_frame->tab_mvf; + const int log2_min_pu_size = sps->log2_min_pu_size; + const int log2_min_tu_size = sps->log2_min_tb_size; + const int min_pu_width = sps->min_pu_width; + const int min_tu_width = sps->min_tb_width; + const int ctb_size = 1 << sps->log2_ctb_size; + int i, bs; + + if ((lc->boundary_flags & BOUNDARY_UPPER_TILE) && y0 > 0) { + const RefPicList *rpl_top = (lc->boundary_flags & BOUNDARY_UPPER_SLICE) ? + ff_hevc_get_ref_list(s->cur_frame, x0, y0 - 1) : + s->cur_frame->refPicList; + int yp_pu = (y0 - 1) >> log2_min_pu_size, yq_pu = y0 >> log2_min_pu_size; + int yp_tu = (y0 - 1) >> log2_min_tu_size, yq_tu = y0 >> log2_min_tu_size; + int len = FFMIN(ctb_size, sps->width - x0); + for (i = 0; i < len; i += 4) { + int x_pu = (x0 + i) >> log2_min_pu_size; + int x_tu = (x0 + i) >> log2_min_tu_size; + const MvField *top = &tab_mvf[yp_pu * min_pu_width + x_pu]; + const MvField *curr = &tab_mvf[yq_pu * min_pu_width + x_pu]; + uint8_t top_cbf = l->cbf_luma[yp_tu * min_tu_width + x_tu]; + uint8_t curr_cbf = l->cbf_luma[yq_tu * min_tu_width + x_tu]; + if (curr->pred_flag == PF_INTRA || top->pred_flag == PF_INTRA) + bs = 2; + else if (curr_cbf || top_cbf) + bs = 1; + else + bs = boundary_strength(s, curr, top, rpl_top); + l->horizontal_bs[((x0 + i) + y0 * l->bs_width) >> 2] = bs; + } + } + + if ((lc->boundary_flags & BOUNDARY_LEFT_TILE) && x0 > 0) { + const RefPicList *rpl_left = (lc->boundary_flags & BOUNDARY_LEFT_SLICE) ? + ff_hevc_get_ref_list(s->cur_frame, x0 - 1, y0) : + s->cur_frame->refPicList; + int xp_pu = (x0 - 1) >> log2_min_pu_size, xq_pu = x0 >> log2_min_pu_size; + int xp_tu = (x0 - 1) >> log2_min_tu_size, xq_tu = x0 >> log2_min_tu_size; + int len = FFMIN(ctb_size, sps->height - y0); + for (i = 0; i < len; i += 4) { + int y_pu = (y0 + i) >> log2_min_pu_size; + int y_tu = (y0 + i) >> log2_min_tu_size; + const MvField *left = &tab_mvf[y_pu * min_pu_width + xp_pu]; + const MvField *curr = &tab_mvf[y_pu * min_pu_width + xq_pu]; + uint8_t left_cbf = l->cbf_luma[y_tu * min_tu_width + xp_tu]; + uint8_t curr_cbf = l->cbf_luma[y_tu * min_tu_width + xq_tu]; + if (curr->pred_flag == PF_INTRA || left->pred_flag == PF_INTRA) + bs = 2; + else if (curr_cbf || left_cbf) + bs = 1; + else + bs = boundary_strength(s, curr, left, rpl_left); + l->vertical_bs[(x0 + (y0 + i) * l->bs_width) >> 2] = bs; + } + } +} + #undef LUMA #undef CB #undef CR diff --git a/libavcodec/hevc/hevcdec.c b/libavcodec/hevc/hevcdec.c index 2ab6b3e2c0..b572d8dd11 100644 --- a/libavcodec/hevc/hevcdec.c +++ b/libavcodec/hevc/hevcdec.c @@ -2733,7 +2733,9 @@ static void hls_decode_neighbour(HEVCLocalContext *lc, int ctb_addr_rs = pps->ctb_addr_ts_to_rs[ctb_addr_ts]; int ctb_addr_in_slice = ctb_addr_rs - s->sh.slice_addr; - l->tab_slice_address[ctb_addr_rs] = s->sh.slice_addr; + /* the tile-parallel path pre-fills this serially, workers only read it */ + if (!lc->tile_bs_defer) + l->tab_slice_address[ctb_addr_rs] = s->sh.slice_addr; if (pps->entropy_coding_sync_enabled_flag) { if (x_ctb == 0 && (y_ctb & (ctb_size - 1)) == 0) @@ -3070,6 +3072,121 @@ static int hls_slice_data_wpp(HEVCContext *s, const H2645NAL *nal) return res; } +static int hls_decode_entry_tile(AVCodecContext *avctx, void *hevc_lclist, + int job, int thread) +{ + HEVCLocalContext *lc = &((HEVCLocalContext*)hevc_lclist)[thread]; + const HEVCContext *const s = lc->parent; + const HEVCLayerContext *const l = &s->layers[s->cur_layer]; + const HEVCPPS *const pps = s->pps; + const HEVCSPS *const sps = pps->sps; + const uint8_t *data = s->data + s->sh.offset[job]; + const size_t data_size = s->sh.size[job]; + /* the slice covers every tile, so job and tile are the same index */ + int ctb_addr_ts = pps->ctb_addr_rs_to_ts[pps->row_bd[job / pps->num_tile_columns] * sps->ctb_width + + pps->col_bd[job % pps->num_tile_columns]]; + int more_data = 1, ret; + + lc->tile_bs_defer = 1; + lc->tu.cu_qp_offset_cb = 0; + lc->tu.cu_qp_offset_cr = 0; + /* hls_decode_neighbour() skips this for the first CTB of the picture */ + lc->end_of_tiles_x = pps->col_bd[job % pps->num_tile_columns + 1] << sps->log2_ctb_size; + + while (more_data && ctb_addr_ts < sps->ctb_size && + pps->tile_id[ctb_addr_ts] == job) { + int ctb_addr_rs = pps->ctb_addr_ts_to_rs[ctb_addr_ts]; + int x_ctb = (ctb_addr_rs % sps->ctb_width) << sps->log2_ctb_size; + int y_ctb = (ctb_addr_rs / sps->ctb_width) << sps->log2_ctb_size; + + hls_decode_neighbour(lc, l, pps, sps, x_ctb, y_ctb, ctb_addr_ts); + + ret = ff_hevc_cabac_init(lc, pps, ctb_addr_ts, data, data_size, 1); + if (ret < 0) + return ret; + + hls_sao_param(lc, l, pps, sps, + x_ctb >> sps->log2_ctb_size, y_ctb >> sps->log2_ctb_size); + + l->deblock[ctb_addr_rs].beta_offset = s->sh.beta_offset; + l->deblock[ctb_addr_rs].tc_offset = s->sh.tc_offset; + l->filter_slice_edges[ctb_addr_rs] = s->sh.slice_loop_filter_across_slices_enabled_flag; + + more_data = hls_coding_quadtree(lc, l, pps, sps, x_ctb, y_ctb, sps->log2_ctb_size, 0); + if (more_data < 0) + return more_data; + ctb_addr_ts++; + } + return ctb_addr_ts; +} + +static int hls_slice_data_tiles(HEVCContext *s, const H2645NAL *nal) +{ + const HEVCPPS *const pps = s->pps; + const HEVCSPS *const sps = pps->sps; + const HEVCLayerContext *const l = &s->layers[s->cur_layer]; + const int ctb_size = 1 << sps->log2_ctb_size; + const int nb_tiles = s->sh.num_entry_point_offsets + 1; + const int start_ts = pps->ctb_addr_rs_to_ts[s->sh.slice_ctb_addr_rs]; + int res = 0, ctb_addr_ts, x_ctb = 0, y_ctb = 0; + int *ret; + + res = alloc_local_ctxs(s); + if (res < 0) + return res; + + res = slice_substreams_init(s, nal); + if (res < 0) + return res; + + for (unsigned i = 1; i < s->nb_local_ctx; i++) { + s->local_ctx[i].first_qp_group = 1; + s->local_ctx[i].qp_y = s->local_ctx[0].qp_y; + } + + for (ctb_addr_ts = start_ts; ctb_addr_ts < sps->ctb_size; ctb_addr_ts++) + l->tab_slice_address[pps->ctb_addr_ts_to_rs[ctb_addr_ts]] = s->sh.slice_addr; + + ret = av_calloc(nb_tiles, sizeof(*ret)); + if (!ret) + return AVERROR(ENOMEM); + s->avctx->execute2(s->avctx, hls_decode_entry_tile, s->local_ctx, ret, nb_tiles); + for (int i = 0; i < nb_tiles; i++) + if (ret[i] < 0) + res = ret[i]; + av_free(ret); + + for (unsigned i = 0; i < s->nb_local_ctx; i++) + s->local_ctx[i].tile_bs_defer = 0; + + if (res < 0) + return res; + + if (pps->loop_filter_across_tiles_enabled_flag && + !s->sh.disable_deblocking_filter_flag) { + for (ctb_addr_ts = start_ts; ctb_addr_ts < sps->ctb_size; ctb_addr_ts++) { + int ctb_addr_rs = pps->ctb_addr_ts_to_rs[ctb_addr_ts]; + x_ctb = (ctb_addr_rs % sps->ctb_width) << sps->log2_ctb_size; + y_ctb = (ctb_addr_rs / sps->ctb_width) << sps->log2_ctb_size; + hls_decode_neighbour(&s->local_ctx[0], l, pps, sps, x_ctb, y_ctb, ctb_addr_ts); + if (s->local_ctx[0].boundary_flags & (BOUNDARY_LEFT_TILE | BOUNDARY_UPPER_TILE)) + ff_hevc_tile_boundary_bs(&s->local_ctx[0], l, pps, x_ctb, y_ctb); + } + } + + for (ctb_addr_ts = start_ts; ctb_addr_ts < sps->ctb_size; ctb_addr_ts++) { + int ctb_addr_rs = pps->ctb_addr_ts_to_rs[ctb_addr_ts]; + x_ctb = (ctb_addr_rs % sps->ctb_width) << sps->log2_ctb_size; + y_ctb = (ctb_addr_rs / sps->ctb_width) << sps->log2_ctb_size; + hls_decode_neighbour(&s->local_ctx[0], l, pps, sps, x_ctb, y_ctb, ctb_addr_ts); + ff_hevc_hls_filters(&s->local_ctx[0], l, pps, x_ctb, y_ctb, ctb_size); + } + if (x_ctb + ctb_size >= sps->width && y_ctb + ctb_size >= sps->height) + ff_hevc_hls_filter(&s->local_ctx[0], l, pps, x_ctb, y_ctb, ctb_size); + + return sps->ctb_size; +} + static int decode_slice_data(HEVCContext *s, const HEVCLayerContext *l, const H2645NAL *nal, GetBitContext *gb) { @@ -3124,6 +3241,14 @@ static int decode_slice_data(HEVCContext *s, const HEVCLayerContext *l, pps->num_tile_rows == 1 && pps->num_tile_columns == 1) return hls_slice_data_wpp(s, nal); + if (s->avctx->active_thread_type == FF_THREAD_SLICE && + s->sh.num_entry_point_offsets > 0 && + pps->tiles_enabled_flag && + !pps->entropy_coding_sync_enabled_flag && + s->sh.first_slice_in_pic_flag && + s->sh.num_entry_point_offsets + 1 == pps->num_tile_rows * pps->num_tile_columns) + return hls_slice_data_tiles(s, nal); + return hls_decode_entry(s, gb); } diff --git a/libavcodec/hevc/hevcdec.h b/libavcodec/hevc/hevcdec.h index 9e5acbd767..703b8df1af 100644 --- a/libavcodec/hevc/hevcdec.h +++ b/libavcodec/hevc/hevcdec.h @@ -444,6 +444,9 @@ typedef struct HEVCLocalContext { * of the deblocking filter */ int boundary_flags; + /* decoding tiles in parallel: defer tile-boundary BS to a serial pass */ + int tile_bs_defer; + // an array of these structs is used for per-thread state - pad its size // to avoid false sharing char padding[128]; @@ -710,6 +713,8 @@ void ff_hevc_set_qPy(HEVCLocalContext *lc, void ff_hevc_deblocking_boundary_strengths(HEVCLocalContext *lc, const HEVCLayerContext *l, const HEVCPPS *pps, int x0, int y0, int log2_trafo_size); +void ff_hevc_tile_boundary_bs(HEVCLocalContext *lc, const HEVCLayerContext *l, + const HEVCPPS *pps, int x0, int y0); int ff_hevc_cu_qp_delta_sign_flag(HEVCLocalContext *lc); int ff_hevc_cu_qp_delta_abs(HEVCLocalContext *lc); int ff_hevc_cu_chroma_qp_offset_flag(HEVCLocalContext *lc); diff --git a/tests/fate/hevc.mak b/tests/fate/hevc.mak index ba5242e0d2..b39358167e 100644 --- a/tests/fate/hevc.mak +++ b/tests/fate/hevc.mak @@ -303,6 +303,15 @@ FATE_HEVC_FFPROBE-$(call DEMDEC, HEVC, HEVC) += fate-hevc-skip-pred-fields fate-hevc-skip-pred-pts: CMD = probeframes -show_entries frame=key_frame,pts,pict_type -skip_pred all -skip_idct all $(TARGET_SAMPLES)/mov/elst_ends_betn_b_and_i.mp4 FATE_HEVC_FFPROBE-$(call DEMDEC, MOV, HEVC) += fate-hevc-skip-pred-pts +# TILES_A conformance picture through the tile-parallel slice path, same output +fate-hevc-tiles-slice-threads: CMD = thread_type=slice threads=4 framecrc -i $(TARGET_SAMPLES)/hevc-conformance/TILES_A_Cisco_2.bit -pix_fmt yuv420p +fate-hevc-tiles-slice-threads: REF = $(SRC_PATH)/tests/ref/fate/hevc-conformance-TILES_A_Cisco_2 +FATE_HEVC-$(call FRAMECRC, HEVC, HEVC, HEVC_PARSER) += fate-hevc-tiles-slice-threads + +# deblocking disabled + loop_filter_across_tiles: tile edges must stay unfiltered +fate-hevc-tiles-nodeblock-slice-threads: CMD = thread_type=slice threads=4 framecrc -i $(TARGET_SAMPLES)/hevc/tiles_nodeblock.hevc +FATE_HEVC-$(call FRAMECRC, HEVC, HEVC, HEVC_PARSER) += fate-hevc-tiles-nodeblock-slice-threads + fate-hevc-cabac-tudepth: CMD = framecrc -i $(TARGET_SAMPLES)/hevc/cbf_cr_cb_TUDepth_4_circle.h265 -pix_fmt yuv444p FATE_HEVC-$(call FRAMECRC, HEVC, HEVC) += fate-hevc-cabac-tudepth diff --git a/tests/ref/fate/hevc-tiles-nodeblock-slice-threads b/tests/ref/fate/hevc-tiles-nodeblock-slice-threads new file mode 100644 index 0000000000..41cd7f1862 --- /dev/null +++ b/tests/ref/fate/hevc-tiles-nodeblock-slice-threads @@ -0,0 +1,20 @@ +#tb 0: 1/30 +#media_type 0: video +#codec_id 0: rawvideo +#dimensions 0: 832x480 +#sar 0: 0/1 +0, 0, 0, 1, 599040, 0x0c2d0ff4 +0, 1, 1, 1, 599040, 0xbb240afc +0, 2, 2, 1, 599040, 0x2a587dd1 +0, 3, 3, 1, 599040, 0x63d9f737 +0, 4, 4, 1, 599040, 0xe62185e1 +0, 5, 5, 1, 599040, 0x3640fd36 +0, 6, 6, 1, 599040, 0x71a7f607 +0, 7, 7, 1, 599040, 0xf7a797ae +0, 8, 8, 1, 599040, 0x743eddfb +0, 9, 9, 1, 599040, 0x2e1d612d +0, 10, 10, 1, 599040, 0x97d2b583 +0, 11, 11, 1, 599040, 0x6186fbcc +0, 12, 12, 1, 599040, 0xd963b62f +0, 13, 13, 1, 599040, 0x47b22c72 +0, 14, 14, 1, 599040, 0x6a73ff19 -- To stop receiving notification emails like this one, please contact [email protected]. _______________________________________________ ffmpeg-cvslog mailing list -- [email protected] To unsubscribe send an email to [email protected]
