Commit 4a1378d726 for aom

commit 4a1378d726fbe71d1ae9b59be0e76bbf48d1f396
Author: Yuan Tong <tongyuan200097@gmail.com>
Date:   Thu Sep 24 23:22:37 2026 +0800

    Allow upscaling without forcing keyframes for GOOD_QUALITY

    The remaining blockers after
    https://aomedia-review.googlesource.com/c/aom/+/216241 are only 2 bugs
    in the heavy motion search routines.

    They can be detected and verified by running the updated test case using
    `test_libaom --gtest_filter=*AV1ResolutionChange.RandomInput*` under
    ASAN.

    Change-Id: Iab55db677c793092fe2eb1bcf640d4534de08dd4

diff --git a/av1/av1_cx_iface.c b/av1/av1_cx_iface.c
index 030da6e1cc..f3d14e1328 100644
--- a/av1/av1_cx_iface.c
+++ b/av1/av1_cx_iface.c
@@ -1672,8 +1672,8 @@ static aom_codec_err_t encoder_set_config(aom_codec_alg_priv_t *ctx,
     if (cfg->g_lag_in_frames > 1 || cfg->g_pass != AOM_RC_ONE_PASS)
       ERROR("Cannot change width or height after initialization");
     // Note: function encoder_set_config() is allowed to be called multiple
-    // times. In single-pass realtime mode without lookahead (g_lag_in_frames ==
-    // 0), and with the maximum frame size declared up front via
+    // times. In one-pass mode without lookahead (g_lag_in_frames == 0),
+    // and with the maximum frame size declared up front via
     // g_forced_max_frame_width/height, reference frame scaling allows upscaling
     // up to 16x and downscaling by up to 2x without forcing a keyframe. The
     // forced maximum frame size is required because the internal buffers are
@@ -1688,8 +1688,7 @@ static aom_codec_err_t encoder_set_config(aom_codec_alg_priv_t *ctx,
     // actual coded frame size.
     const bool allow_ref_scaled_upscale =
         cfg->g_forced_max_frame_width && cfg->g_forced_max_frame_height &&
-        ctx->oxcf.mode == REALTIME && cfg->g_pass == AOM_RC_ONE_PASS &&
-        cfg->g_lag_in_frames == 0;
+        cfg->g_pass == AOM_RC_ONE_PASS && cfg->g_lag_in_frames == 0;
     if (ctx->ppi->cpi->svc.number_spatial_layers == 1 &&
         ctx->ppi->cpi->last_coded_width && ctx->ppi->cpi->last_coded_height &&
         (!valid_ref_frame_size(ctx->ppi->cpi->last_coded_width,
diff --git a/av1/encoder/motion_search_facade.c b/av1/encoder/motion_search_facade.c
index 5d60d9c14a..57a41612e5 100644
--- a/av1/encoder/motion_search_facade.c
+++ b/av1/encoder/motion_search_facade.c
@@ -692,9 +692,19 @@ int av1_joint_motion_search(const AV1_COMP *cpi, MACROBLOCK *x,
     }

     // Do sub-pixel compound motion search on the current reference frame.
+    const struct scale_factors *orig_sf = NULL;
     if (id) {
       orig_yv12 = xd->plane[plane].pre[0];
       xd->plane[plane].pre[0] = xd->plane[plane].pre[id];
+      // Sub-pixel motion search works on raw reference frames that may not
+      // match the resolution of the current frame, so block_ref_scale_factors
+      // must be kept in sync with the frame data in order to perform the
+      // motion search correctly.
+      // Full-pixel motion search works on scaled reference frames, so it
+      // doesn't need to update block_ref_scale_factors when swapping in
+      // scaled_ref_frame.
+      orig_sf = xd->block_ref_scale_factors[0];
+      xd->block_ref_scale_factors[0] = xd->block_ref_scale_factors[id];
     }

     if (cpi->common.features.cur_frame_force_integer_mv) {
@@ -734,7 +744,10 @@ int av1_joint_motion_search(const AV1_COMP *cpi, MACROBLOCK *x,
     }

     // Restore the pointer to the first prediction buffer.
-    if (id) xd->plane[plane].pre[0] = orig_yv12;
+    if (id) {
+      xd->plane[plane].pre[0] = orig_yv12;
+      xd->block_ref_scale_factors[0] = orig_sf;
+    }
     if (bestsme < last_besterr[id]) {
       cur_mv[id] = best_mv;
       last_besterr[id] = bestsme;
@@ -834,6 +847,12 @@ int av1_compound_single_motion_search(const AV1_COMP *cpi, MACROBLOCK *x,
     }
   }

+  const struct scale_factors *orig_sf = NULL;
+  if (ref_idx) {
+    orig_sf = xd->block_ref_scale_factors[0];
+    xd->block_ref_scale_factors[0] = xd->block_ref_scale_factors[ref_idx];
+  }
+
   if (cpi->common.features.cur_frame_force_integer_mv) {
     convert_fullmv_to_mv(&best_mv);
   }
@@ -856,7 +875,10 @@ int av1_compound_single_motion_search(const AV1_COMP *cpi, MACROBLOCK *x,
   }

   // Restore the pointer to the first unscaled prediction buffer.
-  if (ref_idx) pd->pre[0] = orig_yv12;
+  if (ref_idx) {
+    pd->pre[0] = orig_yv12;
+    xd->block_ref_scale_factors[0] = orig_sf;
+  }

   if (bestsme < INT_MAX) *this_mv = best_mv.as_mv;

diff --git a/test/frame_size_tests.cc b/test/frame_size_tests.cc
index 96d544880e..97bbc59ca5 100644
--- a/test/frame_size_tests.cc
+++ b/test/frame_size_tests.cc
@@ -223,24 +223,17 @@ TEST_P(AV1ResolutionChange, RandomInput) {
       while ((pkt = aom_codec_get_cx_data(enc.get(), &iter)) != nullptr) {
         ASSERT_EQ(pkt->kind, AOM_CODEC_CX_FRAME_PKT);
         // All the resolution changes above are within the reference frame
-        // scaling limits (up to 16x up and 2x down). In single pass realtime
-        // mode without lookahead, and with the maximum frame size declared up
-        // front via g_forced_max_frame_width/height, such changes are coded as
-        // inter frames that scale their references, so only the very first
-        // frame is a keyframe. Other modes force a keyframe on every
-        // resolution change.
-        const bool scales_references = usage_ == AOM_USAGE_REALTIME;
+        // scaling limits (up to 16x up and 2x down). In one pass mode
+        // without lookahead, and with the maximum frame size declared up front
+        // via g_forced_max_frame_width/height, such changes are coded as inter
+        // frames that scale their references, so only the very first frame is
+        // a keyframe. All intra mode codes every frame as a keyframe.
         if (usage_ == AOM_USAGE_ALL_INTRA || frame_count == 0) {
           EXPECT_NE(pkt->data.frame.flags & AOM_FRAME_IS_KEY, 0u)
               << "frame " << frame_count;
-        } else if (i == 0) {
-          if (scales_references) {
-            EXPECT_EQ(pkt->data.frame.flags & AOM_FRAME_IS_KEY, 0u)
-                << "frame " << frame_count;
-          } else {
-            EXPECT_NE(pkt->data.frame.flags & AOM_FRAME_IS_KEY, 0u)
-                << "frame " << frame_count;
-          }
+        } else {
+          EXPECT_EQ(pkt->data.frame.flags & AOM_FRAME_IS_KEY, 0u)
+              << "frame " << frame_count;
         }
         frame_count++;
       }