Skip to content

Commit c4c1a8a

Browse files
Optimize TCQ with block-level AVX2 dispatch
Previously, TCQ evaluated coefficients along anti-diagonal scan lines in a portable C loop (trellis_loop_diagonal_st8), invoking individual fine-grained RTCD SIMD functions per coefficient. This fine-grained dispatch structure incurred significant overhead: - Repeated RTCD function calls and indirect branches on every coeff. - Register spills to memory structures across subroutine boundaries. - Optimization barriers preventing compiler vector scheduling across the full diagonal loop. This patch introduces a coarse-grained, block-level RTCD entry point: av2_trellis_loop_diagonal_st8() with AVX2 specialization (av2_trellis_loop_diagonal_st8_avx2). Updates are now executed contiguously within AVX2 registers without exiting to scalar C code per coefficient. Unit tests are added to compare the the new and original implementation. We can see that the TCQ function becomes 3-10% faster. ===================================================================== TX Size Pure C (us) Base AVX2 (us) Patch AVX2 (us) vs Base AVX2 --------------------------------------------------------------- 4x4 110313 45646 44360 1.03x (+2.8%) 8x8 369747 102636 96537 1.06x (+5.9%) 16x16 343618 81354 74201 1.10x (+8.8%) 32x32 268794 60375 54646 1.10x (+9.5%) 4x8 195707 64939 61781 1.05x (+4.9%) 8x4 201683 65227 61872 1.05x (+5.1%) 8x16 175484 44150 40669 1.09x (+7.9%) 16x8 180465 44455 40943 1.09x (+7.9%) 16x32 134924 30975 28011 1.11x (+9.6%) 32x16 134311 30946 28192 1.10x (+8.9%) 4x16 364581 103343 96494 1.07x (+6.6%) 16x4 390933 104326 97702 1.07x (+6.3%) 8x32 343005 81586 74379 1.10x (+8.8%) 32x8 352547 81916 74820 1.09x (+8.7%) ===================================================================== Change-Id: Ib14c7e995e37e4c0af0750af50c78db3d2adb3b5
1 parent 735d104 commit c4c1a8a

4 files changed

Lines changed: 777 additions & 67 deletions

File tree

‎av2/common/av2_rtcd_defs.pl‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -246,6 +246,8 @@ ()
246246
specialize qw/av2_get_coeff_ctx avx2/;
247247
add_proto qw/void av2_update_nbr_diagonal/, "struct tcq_ctx_t *tcq_ctx, int row, int col, int bwl";
248248
specialize qw/av2_update_nbr_diagonal avx2/;
249+
add_proto qw/void av2_trellis_loop_diagonal_st8/, "const struct tcq_param_t *p, int scan_hi, int scan_lo, struct tcq_ctx_t *tcq_ctx, struct tcq_node_t *trellis";
250+
specialize qw/av2_trellis_loop_diagonal_st8 avx2/;
249251

250252
# fdct functions
251253

‎av2/encoder/trellis_quant.c‎

Lines changed: 37 additions & 35 deletions
Original file line numberDiff line numberDiff line change
@@ -935,10 +935,12 @@ static AVM_INLINE int get_diag_ctx(int lf, int blk_pos, int scan_pos, int bwl) {
935935
return diag_ctx;
936936
}
937937

938-
// TCQ 8-state for a 2D luma block.
939-
static void trellis_loop_diagonal_st8(const tcq_param_t *p, int scan_hi,
940-
int scan_lo, tcq_ctx_t *tcq_ctx,
941-
tcq_node_t *trellis) {
938+
// TCQ 8-state for a 2D luma block. Dispatch this whole loop once per block so
939+
// SIMD implementations do not pay indirect-call overhead for every kernel at
940+
// every coefficient.
941+
void av2_trellis_loop_diagonal_st8_c(const tcq_param_t *p, int scan_hi,
942+
int scan_lo, tcq_ctx_t *tcq_ctx,
943+
tcq_node_t *trellis) {
942944
int plane = p->plane;
943945
int log_scale = p->log_scale;
944946
int try_eob = p->sharpness == 0;
@@ -985,35 +987,35 @@ static void trellis_loop_diagonal_st8(const tcq_param_t *p, int scan_hi,
985987

986988
// Get coeff contexts
987989
tcq_coeff_ctx_t coeff_ctx;
988-
av2_get_coeff_ctx(tcq_ctx, col, &coeff_ctx);
990+
av2_get_coeff_ctx_c(tcq_ctx, col, &coeff_ctx);
989991
coeff_ctx.coef_eob = get_lower_levels_ctx_eob(bwl, height, scan_pos);
990992
int eob_rate = block_eob_rate[scan_pos];
991993
tcq_rate_t rd;
992994

993995
if (pqData.orig_qIdx < 2) {
994-
av2_pre_quant_q1(tcoeff[blk_pos], &pqData, quant, tempdqv, log_scale,
995-
scan_pos);
996-
av2_get_rate_dist_def_luma_q1(p, &pqData, &coeff_ctx, blk_pos, diag_ctx,
997-
eob_rate, &rd);
998-
av2_decide_states_q1(prev_decision, &rd, &pqData, lf, try_eob, rdmult,
999-
decision);
996+
av2_pre_quant_q1_c(tcoeff[blk_pos], &pqData, quant, tempdqv, log_scale,
997+
scan_pos);
998+
av2_get_rate_dist_def_luma_q1_c(p, &pqData, &coeff_ctx, blk_pos,
999+
diag_ctx, eob_rate, &rd);
1000+
av2_decide_states_q1_c(prev_decision, &rd, &pqData, lf, try_eob, rdmult,
1001+
decision);
10001002
} else {
1001-
av2_pre_quant(tcoeff[blk_pos], &pqData, quant, tempdqv, log_scale,
1002-
scan_pos);
1003-
av2_get_rate_dist_def_luma(p, &pqData, &coeff_ctx, blk_pos, diag_ctx,
1004-
eob_rate, &rd);
1003+
av2_pre_quant_c(tcoeff[blk_pos], &pqData, quant, tempdqv, log_scale,
1004+
scan_pos);
1005+
av2_get_rate_dist_def_luma_c(p, &pqData, &coeff_ctx, blk_pos, diag_ctx,
1006+
eob_rate, &rd);
10051007

1006-
av2_decide_states(prev_decision, &rd, &pqData, lf, try_eob, rdmult,
1007-
decision);
1008+
av2_decide_states_c(prev_decision, &rd, &pqData, lf, try_eob, rdmult,
1009+
decision);
10081010
}
10091011

1010-
av2_update_states(decision, col, tcq_ctx);
1012+
av2_update_states_c(decision, col, tcq_ctx);
10111013

10121014
blk_pos += blk_pos_inc;
10131015
col--;
10141016
row++;
10151017
}
1016-
av2_update_nbr_diagonal(tcq_ctx, row - 1, col + 1, bwl);
1018+
av2_update_nbr_diagonal_c(tcq_ctx, row - 1, col + 1, bwl);
10171019
scan_hi = scan_lo - 1;
10181020
}
10191021
// Handle LF region.
@@ -1041,36 +1043,36 @@ static void trellis_loop_diagonal_st8(const tcq_param_t *p, int scan_hi,
10411043

10421044
// Get coeff contexts
10431045
tcq_coeff_ctx_t coeff_ctx;
1044-
av2_get_coeff_ctx(tcq_ctx, col, &coeff_ctx);
1046+
av2_get_coeff_ctx_c(tcq_ctx, col, &coeff_ctx);
10451047
coeff_ctx.coef_eob = get_lower_levels_ctx_eob(bwl, height, scan_pos);
10461048
int eob_rate = block_eob_rate[scan_pos];
10471049
tcq_rate_t rd;
10481050

10491051
if (pqData.orig_qIdx < 2) {
1050-
av2_pre_quant_q1(tcoeff[blk_pos], &pqData, quant, tempdqv, log_scale,
1051-
scan_pos);
1052-
av2_get_rate_dist_lf_luma_q1(p, &pqData, &coeff_ctx, blk_pos, diag_ctx,
1053-
eob_rate, dc_coeff_sign, &rd);
1054-
av2_decide_states_q1(prev_decision, &rd, &pqData, lf, try_eob, rdmult,
1055-
decision);
1052+
av2_pre_quant_q1_c(tcoeff[blk_pos], &pqData, quant, tempdqv, log_scale,
1053+
scan_pos);
1054+
av2_get_rate_dist_lf_luma_q1_c(p, &pqData, &coeff_ctx, blk_pos,
1055+
diag_ctx, eob_rate, dc_coeff_sign, &rd);
1056+
av2_decide_states_q1_c(prev_decision, &rd, &pqData, lf, try_eob, rdmult,
1057+
decision);
10561058
} else {
10571059
// Calculate rate and distortion.
1058-
av2_pre_quant(tcoeff[blk_pos], &pqData, quant, tempdqv, log_scale,
1059-
scan_pos);
1060-
av2_get_rate_dist_lf_luma(p, &pqData, &coeff_ctx, blk_pos, diag_ctx,
1061-
eob_rate, dc_coeff_sign, &rd);
1062-
av2_decide_states(prev_decision, &rd, &pqData, lf, try_eob, rdmult,
1063-
decision);
1060+
av2_pre_quant_c(tcoeff[blk_pos], &pqData, quant, tempdqv, log_scale,
1061+
scan_pos);
1062+
av2_get_rate_dist_lf_luma_c(p, &pqData, &coeff_ctx, blk_pos, diag_ctx,
1063+
eob_rate, dc_coeff_sign, &rd);
1064+
av2_decide_states_c(prev_decision, &rd, &pqData, lf, try_eob, rdmult,
1065+
decision);
10641066
}
10651067

1066-
av2_update_states(decision, col, tcq_ctx);
1068+
av2_update_states_c(decision, col, tcq_ctx);
10671069

10681070
blk_pos += blk_pos_inc;
10691071
col--;
10701072
row++;
10711073
}
10721074
if (scan_hi != 0) {
1073-
av2_update_nbr_diagonal(tcq_ctx, row - 1, col + 1, bwl);
1075+
av2_update_nbr_diagonal_c(tcq_ctx, row - 1, col + 1, bwl);
10741076
}
10751077
scan_hi = scan_lo - 1;
10761078
}
@@ -1353,7 +1355,7 @@ int av2_trellis_quant(const struct AV2_COMP *cpi, MACROBLOCK *x, int plane,
13531355
// Speed-up version for 2D Luma by exploiting parallelism
13541356
// Process coeffs diagonal-by-diagonal.
13551357
if (scan_hi >= 0) {
1356-
trellis_loop_diagonal_st8(&param, scan_hi, 0, &tcq_ctx, trellis);
1358+
av2_trellis_loop_diagonal_st8(&param, scan_hi, 0, &tcq_ctx, trellis);
13571359
}
13581360

13591361
// find best path

0 commit comments

Comments
 (0)