Commit 4a1378d726 for aom
commit 4a1378d726fbe71d1ae9b59be0e76bbf48d1f396
Author: Yuan Tong <tongyuan200097@gmail.com>
Date: Thu Sep 24 23:22:37 2026 +0800
Allow upscaling without forcing keyframes for GOOD_QUALITY
The remaining blockers after
https://aomedia-review.googlesource.com/c/aom/+/216241 are only 2 bugs
in the heavy motion search routines.
They can be detected and verified by running the updated test case using
`test_libaom --gtest_filter=*AV1ResolutionChange.RandomInput*` under
ASAN.
Change-Id: Iab55db677c793092fe2eb1bcf640d4534de08dd4
diff --git a/av1/av1_cx_iface.c b/av1/av1_cx_iface.c
index 030da6e1cc..f3d14e1328 100644
--- a/av1/av1_cx_iface.c
+++ b/av1/av1_cx_iface.c
@@ -1672,8 +1672,8 @@ static aom_codec_err_t encoder_set_config(aom_codec_alg_priv_t *ctx,
if (cfg->g_lag_in_frames > 1 || cfg->g_pass != AOM_RC_ONE_PASS)
ERROR("Cannot change width or height after initialization");
// Note: function encoder_set_config() is allowed to be called multiple
- // times. In single-pass realtime mode without lookahead (g_lag_in_frames ==
- // 0), and with the maximum frame size declared up front via
+ // times. In one-pass mode without lookahead (g_lag_in_frames == 0),
+ // and with the maximum frame size declared up front via
// g_forced_max_frame_width/height, reference frame scaling allows upscaling
// up to 16x and downscaling by up to 2x without forcing a keyframe. The
// forced maximum frame size is required because the internal buffers are
@@ -1688,8 +1688,7 @@ static aom_codec_err_t encoder_set_config(aom_codec_alg_priv_t *ctx,
// actual coded frame size.
const bool allow_ref_scaled_upscale =
cfg->g_forced_max_frame_width && cfg->g_forced_max_frame_height &&
- ctx->oxcf.mode == REALTIME && cfg->g_pass == AOM_RC_ONE_PASS &&
- cfg->g_lag_in_frames == 0;
+ cfg->g_pass == AOM_RC_ONE_PASS && cfg->g_lag_in_frames == 0;
if (ctx->ppi->cpi->svc.number_spatial_layers == 1 &&
ctx->ppi->cpi->last_coded_width && ctx->ppi->cpi->last_coded_height &&
(!valid_ref_frame_size(ctx->ppi->cpi->last_coded_width,
diff --git a/av1/encoder/motion_search_facade.c b/av1/encoder/motion_search_facade.c
index 5d60d9c14a..57a41612e5 100644
--- a/av1/encoder/motion_search_facade.c
+++ b/av1/encoder/motion_search_facade.c
@@ -692,9 +692,19 @@ int av1_joint_motion_search(const AV1_COMP *cpi, MACROBLOCK *x,
}
// Do sub-pixel compound motion search on the current reference frame.
+ const struct scale_factors *orig_sf = NULL;
if (id) {
orig_yv12 = xd->plane[plane].pre[0];
xd->plane[plane].pre[0] = xd->plane[plane].pre[id];
+ // Sub-pixel motion search works on raw reference frames that may not
+ // match the resolution of the current frame, so block_ref_scale_factors
+ // must be kept in sync with the frame data in order to perform the
+ // motion search correctly.
+ // Full-pixel motion search works on scaled reference frames, so it
+ // doesn't need to update block_ref_scale_factors when swapping in
+ // scaled_ref_frame.
+ orig_sf = xd->block_ref_scale_factors[0];
+ xd->block_ref_scale_factors[0] = xd->block_ref_scale_factors[id];
}
if (cpi->common.features.cur_frame_force_integer_mv) {
@@ -734,7 +744,10 @@ int av1_joint_motion_search(const AV1_COMP *cpi, MACROBLOCK *x,
}
// Restore the pointer to the first prediction buffer.
- if (id) xd->plane[plane].pre[0] = orig_yv12;
+ if (id) {
+ xd->plane[plane].pre[0] = orig_yv12;
+ xd->block_ref_scale_factors[0] = orig_sf;
+ }
if (bestsme < last_besterr[id]) {
cur_mv[id] = best_mv;
last_besterr[id] = bestsme;
@@ -834,6 +847,12 @@ int av1_compound_single_motion_search(const AV1_COMP *cpi, MACROBLOCK *x,
}
}
+ const struct scale_factors *orig_sf = NULL;
+ if (ref_idx) {
+ orig_sf = xd->block_ref_scale_factors[0];
+ xd->block_ref_scale_factors[0] = xd->block_ref_scale_factors[ref_idx];
+ }
+
if (cpi->common.features.cur_frame_force_integer_mv) {
convert_fullmv_to_mv(&best_mv);
}
@@ -856,7 +875,10 @@ int av1_compound_single_motion_search(const AV1_COMP *cpi, MACROBLOCK *x,
}
// Restore the pointer to the first unscaled prediction buffer.
- if (ref_idx) pd->pre[0] = orig_yv12;
+ if (ref_idx) {
+ pd->pre[0] = orig_yv12;
+ xd->block_ref_scale_factors[0] = orig_sf;
+ }
if (bestsme < INT_MAX) *this_mv = best_mv.as_mv;
diff --git a/test/frame_size_tests.cc b/test/frame_size_tests.cc
index 96d544880e..97bbc59ca5 100644
--- a/test/frame_size_tests.cc
+++ b/test/frame_size_tests.cc
@@ -223,24 +223,17 @@ TEST_P(AV1ResolutionChange, RandomInput) {
while ((pkt = aom_codec_get_cx_data(enc.get(), &iter)) != nullptr) {
ASSERT_EQ(pkt->kind, AOM_CODEC_CX_FRAME_PKT);
// All the resolution changes above are within the reference frame
- // scaling limits (up to 16x up and 2x down). In single pass realtime
- // mode without lookahead, and with the maximum frame size declared up
- // front via g_forced_max_frame_width/height, such changes are coded as
- // inter frames that scale their references, so only the very first
- // frame is a keyframe. Other modes force a keyframe on every
- // resolution change.
- const bool scales_references = usage_ == AOM_USAGE_REALTIME;
+ // scaling limits (up to 16x up and 2x down). In one pass mode
+ // without lookahead, and with the maximum frame size declared up front
+ // via g_forced_max_frame_width/height, such changes are coded as inter
+ // frames that scale their references, so only the very first frame is
+ // a keyframe. All intra mode codes every frame as a keyframe.
if (usage_ == AOM_USAGE_ALL_INTRA || frame_count == 0) {
EXPECT_NE(pkt->data.frame.flags & AOM_FRAME_IS_KEY, 0u)
<< "frame " << frame_count;
- } else if (i == 0) {
- if (scales_references) {
- EXPECT_EQ(pkt->data.frame.flags & AOM_FRAME_IS_KEY, 0u)
- << "frame " << frame_count;
- } else {
- EXPECT_NE(pkt->data.frame.flags & AOM_FRAME_IS_KEY, 0u)
- << "frame " << frame_count;
- }
+ } else {
+ EXPECT_EQ(pkt->data.frame.flags & AOM_FRAME_IS_KEY, 0u)
+ << "frame " << frame_count;
}
frame_count++;
}