AV1 RT: Add CurvFit mode to nonRD inter

Adding functionality to use CurvFit model for RD estimation in nonRD
mode. It is disabled for now as it produces worse results

Change-Id: I2e6259067bc312d261f0c8b122f64e89de026a7d
diff --git a/av1/encoder/nonrd_pickmode.c b/av1/encoder/nonrd_pickmode.c
index cda8563..20fffbb 100644
--- a/av1/encoder/nonrd_pickmode.c
+++ b/av1/encoder/nonrd_pickmode.c
@@ -34,6 +34,8 @@
 #include "av1/encoder/rdopt.h"
 #include "av1/encoder/reconinter_enc.h"
 
+#define _TMP_USE_CURVFIT_ 0
+
 extern int g_pick_inter_mode_cnt;
 typedef struct {
   uint8_t *data;
@@ -445,20 +447,51 @@
   }
 }
 
-static void model_rd_from_sse(const AV1_COMP *const cpi,
-                              const MACROBLOCK *const x, BLOCK_SIZE plane_bsize,
-                              int plane, int64_t sse, int num_samples,
-                              int *rate, int64_t *dist) {
-  (void)num_samples;
+#if _TMP_USE_CURVFIT_
+static void model_rd_with_curvfit(const AV1_COMP *const cpi,
+                                  const MACROBLOCK *const x,
+                                  BLOCK_SIZE plane_bsize, int plane,
+                                  int64_t sse, int num_samples, int *rate,
+                                  int64_t *dist) {
   (void)cpi;
+  (void)plane_bsize;
   const MACROBLOCKD *const xd = &x->e_mbd;
   const struct macroblockd_plane *const pd = &xd->plane[plane];
   const int dequant_shift = (is_cur_buf_hbd(xd)) ? xd->bd - 5 : 3;
+  const int qstep = AOMMAX(pd->dequant_Q3[1] >> dequant_shift, 1);
 
-  av1_model_rd_from_var_lapndz(sse, num_pels_log2_lookup[plane_bsize],
-                               pd->dequant_Q3[1] >> dequant_shift, rate, dist);
-  *dist <<= 4;
+  if (sse == 0) {
+    if (rate) *rate = 0;
+    if (dist) *dist = 0;
+    return;
+  }
+  aom_clear_system_state();
+  const double sse_norm = (double)sse / num_samples;
+  const double qstepsqr = (double)qstep * qstep;
+  const double xqr = log2(sse_norm / qstepsqr);
+
+  double rate_f, dist_by_sse_norm_f;
+  av1_model_rd_curvfit(plane_bsize, sse_norm, xqr, &rate_f,
+                       &dist_by_sse_norm_f);
+
+  const double dist_f = dist_by_sse_norm_f * sse_norm;
+  int rate_i = (int)(AOMMAX(0.0, rate_f * num_samples) + 0.5);
+  int64_t dist_i = (int64_t)(AOMMAX(0.0, dist_f * num_samples) + 0.5);
+  aom_clear_system_state();
+
+  // Check if skip is better
+  if (rate_i == 0) {
+    dist_i = sse << 4;
+  } else if (RDCOST(x->rdmult, rate_i, dist_i) >=
+             RDCOST(x->rdmult, 0, sse << 4)) {
+    rate_i = 0;
+    dist_i = sse << 4;
+  }
+
+  if (rate) *rate = rate_i;
+  if (dist) *dist = dist_i;
 }
+#endif
 
 static void model_rd_for_sb(const AV1_COMP *const cpi, BLOCK_SIZE bsize,
                             MACROBLOCK *x, MACROBLOCKD *xd, int plane_from,
@@ -500,9 +533,14 @@
                     bh);
     }
     sse = ROUND_POWER_OF_TWO(sse, (xd->bd - 8) * 2);
-
-    model_rd_from_sse(cpi, x, plane_bsize, plane, sse, bw * bh, &rate, &dist);
-
+#if _TMP_USE_CURVFIT_
+    model_rd_with_curvfit(cpi, x, plane_bsize, plane, sse, bw * bh, &rate,
+                          &dist);
+#else
+    (void)cpi;
+    rate = INT_MAX;  // this will be overwritten later with block_yrd
+    dist = INT_MAX;
+#endif
     if (plane == 0) x->pred_sse[ref] = (unsigned int)AOMMIN(sse, UINT_MAX);
 
     total_sse += sse;
@@ -514,7 +552,7 @@
     assert(rate_sum >= 0);
   }
 
-  if (skip_txfm_sb) *skip_txfm_sb = total_sse == 0;
+  if (skip_txfm_sb) *skip_txfm_sb = rate_sum == 0;
   if (skip_sse_sb) *skip_sse_sb = total_sse << 4;
   rate_sum = AOMMIN(rate_sum, INT_MAX);
   *out_rate_sum = (int)rate_sum;
@@ -523,7 +561,7 @@
 
 static void block_yrd(AV1_COMP *cpi, MACROBLOCK *x, int mi_row, int mi_col,
                       RD_STATS *this_rdc, int *skippable, int64_t *sse,
-                      BLOCK_SIZE bsize, TX_SIZE tx_size, int rd_computed) {
+                      BLOCK_SIZE bsize, TX_SIZE tx_size) {
   MACROBLOCKD *xd = &x->e_mbd;
   const struct macroblockd_plane *pd = &xd->plane[0];
   struct macroblock_plane *const p = &x->plane[0];
@@ -544,7 +582,6 @@
 
   (void)mi_row;
   (void)mi_col;
-  (void)rd_computed;
   (void)cpi;
 
   aom_subtract_block(bh, bw, p->src_diff, bw, p->src.buf, p->src.stride,
@@ -597,7 +634,7 @@
       block += step;
     }
   }
-
+  this_rdc->skip = *skippable;
   this_rdc->rate = 0;
   if (*sse < INT64_MAX) {
     *sse = (*sse << 6) >> 2;
@@ -811,7 +848,9 @@
   if (plane == 0) {
     int64_t this_sse = INT64_MAX;
     block_yrd(cpi, x, 0, 0, &this_rdc, &args->skippable, &this_sse, plane_bsize,
-              AOMMIN(tx_size, TX_16X16), 0);
+              AOMMIN(tx_size, TX_16X16));
+  } else {
+    return;
   }
 
   p->src.buf = src_buf_base;
@@ -984,10 +1023,11 @@
     int rate_mv = 0;
     int mode_rd_thresh;
     int mode_index;
+#if !_TMP_USE_CURVFIT_
     int64_t this_sse;
     int is_skippable;
+#endif
     int this_early_term = 0;
-    int rd_computed = 0;
     int skip_this_mv = 0;
     int comp_pred = 0;
     int force_mv_inter_layer = 0;
@@ -1155,7 +1195,6 @@
 
     // TODO(kyslov) For large partition blocks, extra testing needs to be done
 
-    rd_computed = 1;
     model_rd_for_sb(cpi, bsize, x, xd, AOM_PLANE_Y, AOM_PLANE_Y, mi_row, mi_col,
                     &this_rdc.rate, &this_rdc.dist, &this_rdc.skip, NULL, NULL,
                     &sse_y, NULL);
@@ -1164,21 +1203,30 @@
 
     const int skip_ctx = av1_get_skip_context(xd);
     const int skip_cost = x->skip_cost[skip_ctx][1];
+    const int no_skip_cost = x->skip_cost[skip_ctx][0];
 
+#if !_TMP_USE_CURVFIT_
     this_sse = (int64_t)sse_y;
-
     block_yrd(cpi, x, mi_row, mi_col, &this_rdc, &is_skippable, &this_sse,
-              bsize, mi->tx_size, rd_computed);
+              bsize, mi->tx_size);
+#endif
 
-    x->skip = is_skippable;
-    if (is_skippable) {
+    x->skip = this_rdc.skip;
+    if (this_rdc.skip) {
       this_rdc.rate = skip_cost;
     } else {
+#if !_TMP_USE_CURVFIT_
+      // on CurvFit this condition is checked inside curvfit modeling
       if (RDCOST(x->rdmult, this_rdc.rate, this_rdc.dist) >=
-          RDCOST(x->rdmult, 0, this_sse)) {
+          RDCOST(x->rdmult, 0,
+                 this_sse)) {  // this_sse already multiplied by 16 in block_yrd
         x->skip = 1;
         this_rdc.rate = skip_cost;
         this_rdc.dist = this_sse;
+      } else
+#endif
+      {
+        this_rdc.rate += no_skip_cost;
       }
     }