Coverage Report

Created: 2026-09-14 08:00

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/libvpx/vp8/encoder/rdopt.c
Line
Count
Source
1
/*
2
 *  Copyright (c) 2010 The WebM project authors. All Rights Reserved.
3
 *
4
 *  Use of this source code is governed by a BSD-style license
5
 *  that can be found in the LICENSE file in the root of the source
6
 *  tree. An additional intellectual property rights grant can be found
7
 *  in the file PATENTS.  All contributing project authors may
8
 *  be found in the AUTHORS file in the root of the source tree.
9
 */
10
11
#include <assert.h>
12
#include <stdio.h>
13
#include <math.h>
14
#include <limits.h>
15
#include <assert.h>
16
#include "vpx_config.h"
17
#include "vp8_rtcd.h"
18
#include "./vpx_dsp_rtcd.h"
19
#include "encodeframe.h"
20
#include "tokenize.h"
21
#include "treewriter.h"
22
#include "onyx_int.h"
23
#include "modecosts.h"
24
#include "encodeintra.h"
25
#include "pickinter.h"
26
#include "vp8/common/common.h"
27
#include "vp8/common/entropymode.h"
28
#include "vp8/common/reconinter.h"
29
#include "vp8/common/reconintra.h"
30
#include "vp8/common/reconintra4x4.h"
31
#include "vp8/common/findnearmv.h"
32
#include "vp8/common/quant_common.h"
33
#include "encodemb.h"
34
#include "vp8/encoder/quantize.h"
35
#include "vpx_dsp/variance.h"
36
#include "vpx_ports/system_state.h"
37
#include "mcomp.h"
38
#include "rdopt.h"
39
#include "vpx_mem/vpx_mem.h"
40
#include "vp8/common/systemdependent.h"
41
#if CONFIG_TEMPORAL_DENOISING
42
#include "denoising.h"
43
#endif
44
extern void vp8_update_zbin_extra(VP8_COMP *cpi, MACROBLOCK *x);
45
46
1.54M
#define MAXF(a, b) (((a) > (b)) ? (a) : (b))
47
48
typedef struct rate_distortion_struct {
49
  int rate2;
50
  int rate_y;
51
  int rate_uv;
52
  int distortion2;
53
  int distortion_uv;
54
} RATE_DISTORTION;
55
56
typedef struct best_mode_struct {
57
  int yrd;
58
  int rd;
59
  int intra_rd;
60
  MB_MODE_INFO mbmode;
61
  union b_mode_info bmodes[16];
62
  PARTITION_INFO partition;
63
} BEST_MODE;
64
65
static const int auto_speed_thresh[17] = { 1000, 200, 150, 130, 150, 125,
66
                                           120,  115, 115, 115, 115, 115,
67
                                           115,  115, 115, 115, 105 };
68
69
const MB_PREDICTION_MODE vp8_mode_order[MAX_MODES] = {
70
  ZEROMV,    DC_PRED,
71
72
  NEARESTMV, NEARMV,
73
74
  ZEROMV,    NEARESTMV,
75
76
  ZEROMV,    NEARESTMV,
77
78
  NEARMV,    NEARMV,
79
80
  V_PRED,    H_PRED,    TM_PRED,
81
82
  NEWMV,     NEWMV,     NEWMV,
83
84
  SPLITMV,   SPLITMV,   SPLITMV,
85
86
  B_PRED,
87
};
88
89
/* This table determines the search order in reference frame priority order,
90
 * which may not necessarily match INTRA,LAST,GOLDEN,ARF
91
 */
92
const int vp8_ref_frame_order[MAX_MODES] = {
93
  1, 0,
94
95
  1, 1,
96
97
  2, 2,
98
99
  3, 3,
100
101
  2, 3,
102
103
  0, 0, 0,
104
105
  1, 2, 3,
106
107
  1, 2, 3,
108
109
  0,
110
};
111
112
static void fill_token_costs(
113
    int c[BLOCK_TYPES][COEF_BANDS][PREV_COEF_CONTEXTS][MAX_ENTROPY_TOKENS],
114
    const vp8_prob p[BLOCK_TYPES][COEF_BANDS][PREV_COEF_CONTEXTS]
115
151k
                    [ENTROPY_NODES]) {
116
151k
  int i, j, k;
117
118
759k
  for (i = 0; i < BLOCK_TYPES; ++i) {
119
5.46M
    for (j = 0; j < COEF_BANDS; ++j) {
120
19.4M
      for (k = 0; k < PREV_COEF_CONTEXTS; ++k) {
121
        /* check for pt=0 and band > 1 if block type 0
122
         * and 0 if blocktype 1
123
         */
124
14.5M
        if (k == 0 && j > (i == 0)) {
125
4.10M
          vp8_cost_tokens2(c[i][j][k], p[i][j][k], vp8_coef_tree, 2);
126
10.4M
        } else {
127
10.4M
          vp8_cost_tokens(c[i][j][k], p[i][j][k], vp8_coef_tree);
128
10.4M
        }
129
14.5M
      }
130
4.86M
    }
131
607k
  }
132
151k
}
133
134
static const int rd_iifactor[32] = { 4, 4, 3, 2, 1, 0, 0, 0, 0, 0, 0,
135
                                     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
136
                                     0, 0, 0, 0, 0, 0, 0, 0, 0, 0 };
137
138
/* values are now correlated to quantizer */
139
static const int sad_per_bit16lut[QINDEX_RANGE] = {
140
  2,  2,  2,  2,  2,  2,  2,  2,  2,  2,  2,  2,  2,  2,  2,  2,  3,  3,  3,
141
  3,  3,  3,  3,  3,  3,  3,  3,  3,  3,  3,  4,  4,  4,  4,  4,  4,  4,  4,
142
  4,  4,  4,  4,  5,  5,  5,  5,  5,  5,  5,  5,  5,  5,  5,  5,  6,  6,  6,
143
  6,  6,  6,  6,  6,  6,  6,  6,  6,  7,  7,  7,  7,  7,  7,  7,  7,  7,  7,
144
  7,  7,  8,  8,  8,  8,  8,  8,  8,  8,  8,  8,  8,  8,  9,  9,  9,  9,  9,
145
  9,  9,  9,  9,  9,  9,  9,  10, 10, 10, 10, 10, 10, 10, 10, 11, 11, 11, 11,
146
  11, 11, 12, 12, 12, 12, 12, 12, 13, 13, 13, 13, 14, 14
147
};
148
static const int sad_per_bit4lut[QINDEX_RANGE] = {
149
  2,  2,  2,  2,  2,  2,  3,  3,  3,  3,  3,  3,  3,  3,  3,  3,  3,  3,  3,
150
  3,  4,  4,  4,  4,  4,  4,  4,  4,  4,  4,  5,  5,  5,  5,  5,  5,  6,  6,
151
  6,  6,  6,  6,  6,  6,  6,  6,  6,  6,  7,  7,  7,  7,  7,  7,  7,  7,  7,
152
  7,  7,  7,  7,  8,  8,  8,  8,  8,  9,  9,  9,  9,  9,  9,  10, 10, 10, 10,
153
  10, 10, 10, 10, 11, 11, 11, 11, 11, 11, 11, 11, 12, 12, 12, 12, 12, 12, 12,
154
  12, 13, 13, 13, 13, 13, 13, 13, 14, 14, 14, 14, 14, 15, 15, 15, 15, 16, 16,
155
  16, 16, 17, 17, 17, 18, 18, 18, 19, 19, 19, 20, 20, 20,
156
};
157
158
151k
void vp8cx_initialize_me_consts(VP8_COMP *cpi, int QIndex) {
159
151k
  cpi->mb.sadperbit16 = sad_per_bit16lut[QIndex];
160
151k
  cpi->mb.sadperbit4 = sad_per_bit4lut[QIndex];
161
151k
}
162
163
151k
void vp8_initialize_rd_consts(VP8_COMP *cpi, MACROBLOCK *x, int Qvalue) {
164
151k
  int q;
165
151k
  int i;
166
151k
  double capped_q = (Qvalue < 160) ? (double)Qvalue : 160.0;
167
151k
  double rdconst = 2.80;
168
169
151k
  vpx_clear_system_state();
170
171
  /* Further tests required to see if optimum is different
172
   * for key frames, golden frames and arf frames.
173
   */
174
151k
  cpi->RDMULT = (int)(rdconst * (capped_q * capped_q));
175
176
  /* Extend rate multiplier along side quantizer zbin increases */
177
151k
  if (cpi->mb.zbin_over_quant > 0) {
178
29.1k
    double oq_factor;
179
29.1k
    double modq;
180
181
    /* Experimental code using the same basic equation as used for Q above
182
     * The units of cpi->mb.zbin_over_quant are 1/128 of Q bin size
183
     */
184
29.1k
    oq_factor = 1.0 + ((double)0.0015625 * cpi->mb.zbin_over_quant);
185
29.1k
    modq = (int)((double)capped_q * oq_factor);
186
29.1k
    cpi->RDMULT = (int)(rdconst * (modq * modq));
187
29.1k
  }
188
189
151k
  if (cpi->pass == 2 && (cpi->common.frame_type != KEY_FRAME)) {
190
0
    if (cpi->twopass.next_iiratio > 31) {
191
0
      cpi->RDMULT += (cpi->RDMULT * rd_iifactor[31]) >> 4;
192
0
    } else {
193
0
      cpi->RDMULT +=
194
0
          (cpi->RDMULT * rd_iifactor[cpi->twopass.next_iiratio]) >> 4;
195
0
    }
196
0
  }
197
198
151k
  cpi->mb.errorperbit = (cpi->RDMULT / 110);
199
151k
  cpi->mb.errorperbit += (cpi->mb.errorperbit == 0);
200
201
151k
  vp8_set_speed_features(cpi);
202
203
3.18M
  for (i = 0; i < MAX_MODES; ++i) {
204
3.03M
    x->mode_test_hit_counts[i] = 0;
205
3.03M
  }
206
207
151k
  q = (int)pow(Qvalue, 1.25);
208
209
151k
  if (q < 8) q = 8;
210
211
151k
  if (cpi->RDMULT > 1000) {
212
85.7k
    cpi->RDDIV = 1;
213
85.7k
    cpi->RDMULT /= 100;
214
215
1.80M
    for (i = 0; i < MAX_MODES; ++i) {
216
1.71M
      if (cpi->sf.thresh_mult[i] < INT_MAX) {
217
1.63M
        x->rd_threshes[i] = cpi->sf.thresh_mult[i] * q / 100;
218
1.63M
      } else {
219
77.9k
        x->rd_threshes[i] = INT_MAX;
220
77.9k
      }
221
222
1.71M
      cpi->rd_baseline_thresh[i] = x->rd_threshes[i];
223
1.71M
    }
224
85.7k
  } else {
225
66.1k
    cpi->RDDIV = 100;
226
227
1.38M
    for (i = 0; i < MAX_MODES; ++i) {
228
1.32M
      if (cpi->sf.thresh_mult[i] < (INT_MAX / q)) {
229
1.21M
        x->rd_threshes[i] = cpi->sf.thresh_mult[i] * q;
230
1.21M
      } else {
231
103k
        x->rd_threshes[i] = INT_MAX;
232
103k
      }
233
234
1.32M
      cpi->rd_baseline_thresh[i] = x->rd_threshes[i];
235
1.32M
    }
236
66.1k
  }
237
238
151k
  {
239
    /* build token cost array for the type of frame we have now */
240
151k
    FRAME_CONTEXT *l = &cpi->lfc_n;
241
242
151k
    if (cpi->common.refresh_alt_ref_frame) {
243
34.8k
      l = &cpi->lfc_a;
244
117k
    } else if (cpi->common.refresh_golden_frame) {
245
11.5k
      l = &cpi->lfc_g;
246
11.5k
    }
247
248
151k
    fill_token_costs(cpi->mb.token_costs,
249
151k
                     (const vp8_prob(*)[8][3][11])l->coef_probs);
250
    /*
251
    fill_token_costs(
252
        cpi->mb.token_costs,
253
        (const vp8_prob( *)[8][3][11]) cpi->common.fc.coef_probs);
254
    */
255
256
    /* TODO make these mode costs depend on last,alt or gold too.  (jbb) */
257
151k
    vp8_init_mode_costs(cpi);
258
151k
  }
259
151k
}
260
261
60.3k
void vp8_auto_select_speed(VP8_COMP *cpi) {
262
60.3k
  int milliseconds_for_compress = (int)(1000000 / cpi->framerate);
263
264
60.3k
  milliseconds_for_compress =
265
60.3k
      milliseconds_for_compress * (16 - cpi->oxcf.cpu_used) / 16;
266
267
#if 0
268
269
    if (0)
270
    {
271
        FILE *f;
272
273
        f = fopen("speed.stt", "a");
274
        fprintf(f, " %8ld %10ld %10ld %10ld\n",
275
                cpi->common.current_video_frame, cpi->Speed, milliseconds_for_compress, cpi->avg_pick_mode_time);
276
        fclose(f);
277
    }
278
279
#endif
280
281
60.3k
  if (cpi->avg_pick_mode_time < milliseconds_for_compress &&
282
60.3k
      (cpi->avg_encode_time - cpi->avg_pick_mode_time) <
283
60.3k
          milliseconds_for_compress) {
284
60.3k
    if (cpi->avg_pick_mode_time == 0) {
285
2.68k
      cpi->Speed = 4;
286
57.6k
    } else {
287
57.6k
      if (milliseconds_for_compress * 100 < cpi->avg_encode_time * 95) {
288
9
        cpi->Speed += 2;
289
9
        cpi->avg_pick_mode_time = 0;
290
9
        cpi->avg_encode_time = 0;
291
292
9
        if (cpi->Speed > 16) {
293
0
          cpi->Speed = 16;
294
0
        }
295
9
      }
296
297
57.6k
      if (milliseconds_for_compress * 100 >
298
57.6k
          cpi->avg_encode_time * auto_speed_thresh[cpi->Speed]) {
299
57.1k
        cpi->Speed -= 1;
300
57.1k
        cpi->avg_pick_mode_time = 0;
301
57.1k
        cpi->avg_encode_time = 0;
302
303
        /* In real-time mode, cpi->speed is in [4, 16]. */
304
57.1k
        if (cpi->Speed < 4) {
305
57.1k
          cpi->Speed = 4;
306
57.1k
        }
307
57.1k
      }
308
57.6k
    }
309
60.3k
  } else {
310
0
    cpi->Speed += 4;
311
312
0
    if (cpi->Speed > 16) cpi->Speed = 16;
313
314
0
    cpi->avg_pick_mode_time = 0;
315
0
    cpi->avg_encode_time = 0;
316
0
  }
317
60.3k
}
318
319
0
int vp8_block_error_c(short *coeff, short *dqcoeff) {
320
0
  int i;
321
0
  int error = 0;
322
323
0
  for (i = 0; i < 16; ++i) {
324
0
    int this_diff = coeff[i] - dqcoeff[i];
325
0
    error += this_diff * this_diff;
326
0
  }
327
328
0
  return error;
329
0
}
330
331
0
int vp8_mbblock_error_c(MACROBLOCK *mb, int dc) {
332
0
  BLOCK *be;
333
0
  BLOCKD *bd;
334
0
  int i, j;
335
0
  int berror, error = 0;
336
337
0
  for (i = 0; i < 16; ++i) {
338
0
    be = &mb->block[i];
339
0
    bd = &mb->e_mbd.block[i];
340
341
0
    berror = 0;
342
343
0
    for (j = dc; j < 16; ++j) {
344
0
      int this_diff = be->coeff[j] - bd->dqcoeff[j];
345
0
      berror += this_diff * this_diff;
346
0
    }
347
348
0
    error += berror;
349
0
  }
350
351
0
  return error;
352
0
}
353
354
0
int vp8_mbuverror_c(MACROBLOCK *mb) {
355
0
  BLOCK *be;
356
0
  BLOCKD *bd;
357
358
0
  int i;
359
0
  int error = 0;
360
361
0
  for (i = 16; i < 24; ++i) {
362
0
    be = &mb->block[i];
363
0
    bd = &mb->e_mbd.block[i];
364
365
0
    error += vp8_block_error_c(be->coeff, bd->dqcoeff);
366
0
  }
367
368
0
  return error;
369
0
}
370
371
20.4k
int VP8_UVSSE(MACROBLOCK *x) {
372
20.4k
  unsigned char *uptr, *vptr;
373
20.4k
  unsigned char *upred_ptr = (*(x->block[16].base_src) + x->block[16].src);
374
20.4k
  unsigned char *vpred_ptr = (*(x->block[20].base_src) + x->block[20].src);
375
20.4k
  int uv_stride = x->block[16].src_stride;
376
377
20.4k
  unsigned int sse1 = 0;
378
20.4k
  unsigned int sse2 = 0;
379
20.4k
  int mv_row = x->e_mbd.mode_info_context->mbmi.mv.as_mv.row;
380
20.4k
  int mv_col = x->e_mbd.mode_info_context->mbmi.mv.as_mv.col;
381
20.4k
  int offset;
382
20.4k
  int pre_stride = x->e_mbd.pre.uv_stride;
383
384
20.4k
  if (mv_row < 0) {
385
691
    mv_row -= 1;
386
19.7k
  } else {
387
19.7k
    mv_row += 1;
388
19.7k
  }
389
390
20.4k
  if (mv_col < 0) {
391
1.09k
    mv_col -= 1;
392
19.3k
  } else {
393
19.3k
    mv_col += 1;
394
19.3k
  }
395
396
20.4k
  mv_row /= 2;
397
20.4k
  mv_col /= 2;
398
399
20.4k
  offset = (mv_row >> 3) * pre_stride + (mv_col >> 3);
400
20.4k
  uptr = x->e_mbd.pre.u_buffer + offset;
401
20.4k
  vptr = x->e_mbd.pre.v_buffer + offset;
402
403
20.4k
  if ((mv_row | mv_col) & 7) {
404
2.55k
    vpx_sub_pixel_variance8x8(uptr, pre_stride, mv_col & 7, mv_row & 7,
405
2.55k
                              upred_ptr, uv_stride, &sse2);
406
2.55k
    vpx_sub_pixel_variance8x8(vptr, pre_stride, mv_col & 7, mv_row & 7,
407
2.55k
                              vpred_ptr, uv_stride, &sse1);
408
2.55k
    sse2 += sse1;
409
17.8k
  } else {
410
17.8k
    vpx_variance8x8(uptr, pre_stride, upred_ptr, uv_stride, &sse2);
411
17.8k
    vpx_variance8x8(vptr, pre_stride, vpred_ptr, uv_stride, &sse1);
412
17.8k
    sse2 += sse1;
413
17.8k
  }
414
20.4k
  return sse2;
415
20.4k
}
416
417
static int cost_coeffs(MACROBLOCK *mb, BLOCKD *b, int type, ENTROPY_CONTEXT *a,
418
411M
                       ENTROPY_CONTEXT *l) {
419
411M
  int c = !type; /* start at coef 0, unless Y with Y2 */
420
411M
  int eob = (int)(*b->eob);
421
411M
  int pt; /* surrounding block/prev coef predictor */
422
411M
  int cost = 0;
423
411M
  short *qcoeff_ptr = b->qcoeff;
424
425
411M
  VP8_COMBINEENTROPYCONTEXTS(pt, *a, *l);
426
427
411M
  assert(eob <= 16);
428
3.99G
  for (; c < eob; ++c) {
429
3.58G
    const int v = qcoeff_ptr[vp8_default_zig_zag1d[c]];
430
3.58G
    const int t = vp8_dct_value_tokens_ptr[v].Token;
431
3.58G
    cost += mb->token_costs[type][vp8_coef_bands[c]][pt][t];
432
3.58G
    cost += vp8_dct_value_cost_ptr[v];
433
3.58G
    pt = vp8_prev_token_class[t];
434
3.58G
  }
435
436
411M
  if (c < 16) {
437
264M
    cost += mb->token_costs[type][vp8_coef_bands[c]][pt][DCT_EOB_TOKEN];
438
264M
  }
439
440
411M
  pt = (c != !type); /* is eob first coefficient; */
441
411M
  *a = *l = pt;
442
443
411M
  return cost;
444
411M
}
445
446
7.58M
static int vp8_rdcost_mby(MACROBLOCK *mb) {
447
7.58M
  int cost = 0;
448
7.58M
  int b;
449
7.58M
  MACROBLOCKD *x = &mb->e_mbd;
450
7.58M
  ENTROPY_CONTEXT_PLANES t_above, t_left;
451
7.58M
  ENTROPY_CONTEXT *ta;
452
7.58M
  ENTROPY_CONTEXT *tl;
453
454
7.58M
  t_above = *mb->e_mbd.above_context;
455
7.58M
  t_left = *mb->e_mbd.left_context;
456
457
7.58M
  ta = (ENTROPY_CONTEXT *)&t_above;
458
7.58M
  tl = (ENTROPY_CONTEXT *)&t_left;
459
460
128M
  for (b = 0; b < 16; ++b) {
461
121M
    cost += cost_coeffs(mb, x->block + b, PLANE_TYPE_Y_NO_DC,
462
121M
                        ta + vp8_block2above[b], tl + vp8_block2left[b]);
463
121M
  }
464
465
7.58M
  cost += cost_coeffs(mb, x->block + 24, PLANE_TYPE_Y2,
466
7.58M
                      ta + vp8_block2above[24], tl + vp8_block2left[24]);
467
468
7.58M
  return cost;
469
7.58M
}
470
471
7.58M
static void macro_block_yrd(MACROBLOCK *mb, int *Rate, int *Distortion) {
472
7.58M
  int b;
473
7.58M
  MACROBLOCKD *const x = &mb->e_mbd;
474
7.58M
  BLOCK *const mb_y2 = mb->block + 24;
475
7.58M
  BLOCKD *const x_y2 = x->block + 24;
476
7.58M
  short *Y2DCPtr = mb_y2->src_diff;
477
7.58M
  BLOCK *beptr;
478
7.58M
  int d;
479
480
7.58M
  vp8_subtract_mby(mb->src_diff, *(mb->block[0].base_src),
481
7.58M
                   mb->block[0].src_stride, mb->e_mbd.predictor, 16);
482
483
  /* Fdct and building the 2nd order block */
484
68.2M
  for (beptr = mb->block; beptr < mb->block + 16; beptr += 2) {
485
60.6M
    mb->short_fdct8x4(beptr->src_diff, beptr->coeff, 32);
486
60.6M
    *Y2DCPtr++ = beptr->coeff[0];
487
60.6M
    *Y2DCPtr++ = beptr->coeff[16];
488
60.6M
  }
489
490
  /* 2nd order fdct */
491
7.58M
  mb->short_walsh4x4(mb_y2->src_diff, mb_y2->coeff, 8);
492
493
  /* Quantization */
494
128M
  for (b = 0; b < 16; ++b) {
495
121M
    mb->quantize_b(&mb->block[b], &mb->e_mbd.block[b]);
496
121M
  }
497
498
  /* DC predication and Quantization of 2nd Order block */
499
7.58M
  mb->quantize_b(mb_y2, x_y2);
500
501
  /* Distortion */
502
7.58M
  d = vp8_mbblock_error(mb, 1) << 2;
503
7.58M
  d += vp8_block_error(mb_y2->coeff, x_y2->dqcoeff);
504
505
7.58M
  *Distortion = (d >> 4);
506
507
  /* rate */
508
7.58M
  *Rate = vp8_rdcost_mby(mb);
509
7.58M
}
510
511
23.1M
static void copy_predictor(unsigned char *dst, const unsigned char *predictor) {
512
23.1M
  const unsigned int *p = (const unsigned int *)predictor;
513
23.1M
  unsigned int *d = (unsigned int *)dst;
514
23.1M
  d[0] = p[0];
515
23.1M
  d[4] = p[4];
516
23.1M
  d[8] = p[8];
517
23.1M
  d[12] = p[12];
518
23.1M
}
519
static int rd_pick_intra4x4block(MACROBLOCK *x, BLOCK *be, BLOCKD *b,
520
                                 B_PREDICTION_MODE *best_mode,
521
                                 const int *bmode_costs, ENTROPY_CONTEXT *a,
522
                                 ENTROPY_CONTEXT *l,
523
524
                                 int *bestrate, int *bestratey,
525
12.1M
                                 int *bestdistortion) {
526
12.1M
  B_PREDICTION_MODE mode;
527
12.1M
  int best_rd = INT_MAX;
528
12.1M
  int rate = 0;
529
12.1M
  int distortion;
530
531
12.1M
  ENTROPY_CONTEXT ta = *a, tempa = *a;
532
12.1M
  ENTROPY_CONTEXT tl = *l, templ = *l;
533
  /*
534
   * The predictor buffer is a 2d buffer with a stride of 16.  Create
535
   * a temp buffer that meets the stride requirements, but we are only
536
   * interested in the left 4x4 block
537
   * */
538
12.1M
  DECLARE_ALIGNED(16, unsigned char, best_predictor[16 * 4]);
539
12.1M
  DECLARE_ALIGNED(16, short, best_dqcoeff[16]);
540
12.1M
  int dst_stride = x->e_mbd.dst.y_stride;
541
12.1M
  unsigned char *dst = x->e_mbd.dst.y_buffer + b->offset;
542
543
12.1M
  unsigned char *Above = dst - dst_stride;
544
12.1M
  unsigned char *yleft = dst - 1;
545
12.1M
  unsigned char top_left = Above[-1];
546
547
134M
  for (mode = B_DC_PRED; mode <= B_HU_PRED; ++mode) {
548
121M
    int this_rd;
549
121M
    int ratey;
550
551
121M
    rate = bmode_costs[mode];
552
553
121M
    vp8_intra4x4_predict(Above, yleft, dst_stride, mode, b->predictor, 16,
554
121M
                         top_left);
555
121M
    vp8_subtract_b(be, b, 16);
556
121M
    x->short_fdct4x4(be->src_diff, be->coeff, 32);
557
121M
    x->quantize_b(be, b);
558
559
121M
    tempa = ta;
560
121M
    templ = tl;
561
562
121M
    ratey = cost_coeffs(x, b, PLANE_TYPE_Y_WITH_DC, &tempa, &templ);
563
121M
    rate += ratey;
564
121M
    distortion = vp8_block_error(be->coeff, b->dqcoeff) >> 2;
565
566
121M
    this_rd = RDCOST(x->rdmult, x->rddiv, rate, distortion);
567
568
121M
    if (this_rd < best_rd) {
569
23.1M
      *bestrate = rate;
570
23.1M
      *bestratey = ratey;
571
23.1M
      *bestdistortion = distortion;
572
23.1M
      best_rd = this_rd;
573
23.1M
      *best_mode = mode;
574
23.1M
      *a = tempa;
575
23.1M
      *l = templ;
576
23.1M
      copy_predictor(best_predictor, b->predictor);
577
23.1M
      memcpy(best_dqcoeff, b->dqcoeff, 32);
578
23.1M
    }
579
121M
  }
580
12.1M
  b->bmi.as_mode = *best_mode;
581
582
12.1M
  vp8_short_idct4x4llm(best_dqcoeff, best_predictor, 16, dst, dst_stride);
583
584
12.1M
  return best_rd;
585
12.1M
}
586
587
static int rd_pick_intra4x4mby_modes(MACROBLOCK *mb, int *Rate, int *rate_y,
588
1.06M
                                     int *Distortion, int best_rd) {
589
1.06M
  MACROBLOCKD *const xd = &mb->e_mbd;
590
1.06M
  int i;
591
1.06M
  int cost = mb->mbmode_cost[xd->frame_type][B_PRED];
592
1.06M
  int distortion = 0;
593
1.06M
  int tot_rate_y = 0;
594
1.06M
  int64_t total_rd = 0;
595
1.06M
  ENTROPY_CONTEXT_PLANES t_above, t_left;
596
1.06M
  ENTROPY_CONTEXT *ta;
597
1.06M
  ENTROPY_CONTEXT *tl;
598
1.06M
  const int *bmode_costs;
599
600
1.06M
  t_above = *mb->e_mbd.above_context;
601
1.06M
  t_left = *mb->e_mbd.left_context;
602
603
1.06M
  ta = (ENTROPY_CONTEXT *)&t_above;
604
1.06M
  tl = (ENTROPY_CONTEXT *)&t_left;
605
606
1.06M
  intra_prediction_down_copy(xd, xd->dst.y_buffer - xd->dst.y_stride + 16);
607
608
1.06M
  bmode_costs = mb->inter_bmode_costs;
609
610
12.6M
  for (i = 0; i < 16; ++i) {
611
12.1M
    MODE_INFO *const mic = xd->mode_info_context;
612
12.1M
    const int mis = xd->mode_info_stride;
613
12.1M
    B_PREDICTION_MODE best_mode = B_MODE_COUNT;
614
12.1M
    int r = 0, ry = 0, d = 0;
615
616
12.1M
    if (mb->e_mbd.frame_type == KEY_FRAME) {
617
6.18M
      const B_PREDICTION_MODE A = above_block_mode(mic, i, mis);
618
6.18M
      const B_PREDICTION_MODE L = left_block_mode(mic, i);
619
620
6.18M
      bmode_costs = mb->bmode_costs[A][L];
621
6.18M
    }
622
623
12.1M
    total_rd += rd_pick_intra4x4block(
624
12.1M
        mb, mb->block + i, xd->block + i, &best_mode, bmode_costs,
625
12.1M
        ta + vp8_block2above[i], tl + vp8_block2left[i], &r, &ry, &d);
626
627
12.1M
    cost += r;
628
12.1M
    distortion += d;
629
12.1M
    tot_rate_y += ry;
630
631
12.1M
    assert(best_mode != B_MODE_COUNT);
632
12.1M
    mic->bmi[i].as_mode = best_mode;
633
634
12.1M
    if (total_rd >= (int64_t)best_rd) break;
635
12.1M
  }
636
637
1.06M
  if (total_rd >= (int64_t)best_rd) return INT_MAX;
638
639
493k
  *Rate = cost;
640
493k
  *rate_y = tot_rate_y;
641
493k
  *Distortion = distortion;
642
643
493k
  return RDCOST(mb->rdmult, mb->rddiv, cost, distortion);
644
1.06M
}
645
646
static int rd_pick_intra16x16mby_mode(MACROBLOCK *x, int *Rate, int *rate_y,
647
522k
                                      int *Distortion) {
648
522k
  MB_PREDICTION_MODE mode;
649
522k
  MB_PREDICTION_MODE mode_selected = MB_MODE_COUNT;
650
522k
  int rate, ratey;
651
522k
  int distortion;
652
522k
  int best_rd = INT_MAX;
653
522k
  int this_rd;
654
522k
  MACROBLOCKD *xd = &x->e_mbd;
655
656
  /* Y Search for 16x16 intra prediction mode */
657
2.61M
  for (mode = DC_PRED; mode <= TM_PRED; ++mode) {
658
2.08M
    xd->mode_info_context->mbmi.mode = mode;
659
660
2.08M
    vp8_build_intra_predictors_mby_s(xd, xd->dst.y_buffer - xd->dst.y_stride,
661
2.08M
                                     xd->dst.y_buffer - 1, xd->dst.y_stride,
662
2.08M
                                     xd->predictor, 16);
663
664
2.08M
    macro_block_yrd(x, &ratey, &distortion);
665
2.08M
    rate = ratey +
666
2.08M
           x->mbmode_cost[xd->frame_type][xd->mode_info_context->mbmi.mode];
667
668
2.08M
    this_rd = RDCOST(x->rdmult, x->rddiv, rate, distortion);
669
670
2.08M
    if (this_rd < best_rd) {
671
665k
      mode_selected = mode;
672
665k
      best_rd = this_rd;
673
665k
      *Rate = rate;
674
665k
      *rate_y = ratey;
675
665k
      *Distortion = distortion;
676
665k
    }
677
2.08M
  }
678
679
522k
  assert(mode_selected != MB_MODE_COUNT);
680
522k
  xd->mode_info_context->mbmi.mode = mode_selected;
681
522k
  return best_rd;
682
522k
}
683
684
8.41M
static int rd_cost_mbuv(MACROBLOCK *mb) {
685
8.41M
  int b;
686
8.41M
  int cost = 0;
687
8.41M
  MACROBLOCKD *x = &mb->e_mbd;
688
8.41M
  ENTROPY_CONTEXT_PLANES t_above, t_left;
689
8.41M
  ENTROPY_CONTEXT *ta;
690
8.41M
  ENTROPY_CONTEXT *tl;
691
692
8.41M
  t_above = *mb->e_mbd.above_context;
693
8.41M
  t_left = *mb->e_mbd.left_context;
694
695
8.41M
  ta = (ENTROPY_CONTEXT *)&t_above;
696
8.41M
  tl = (ENTROPY_CONTEXT *)&t_left;
697
698
75.6M
  for (b = 16; b < 24; ++b) {
699
67.2M
    cost += cost_coeffs(mb, x->block + b, PLANE_TYPE_UV,
700
67.2M
                        ta + vp8_block2above[b], tl + vp8_block2left[b]);
701
67.2M
  }
702
703
8.41M
  return cost;
704
8.41M
}
705
706
static int rd_inter16x16_uv(VP8_COMP *cpi, MACROBLOCK *x, int *rate,
707
2.84M
                            int *distortion, int fullpixel) {
708
2.84M
  (void)cpi;
709
2.84M
  (void)fullpixel;
710
711
2.84M
  vp8_build_inter16x16_predictors_mbuv(&x->e_mbd);
712
2.84M
  vp8_subtract_mbuv(x->src_diff, x->src.u_buffer, x->src.v_buffer,
713
2.84M
                    x->src.uv_stride, &x->e_mbd.predictor[256],
714
2.84M
                    &x->e_mbd.predictor[320], 8);
715
716
2.84M
  vp8_transform_mbuv(x);
717
2.84M
  vp8_quantize_mbuv(x);
718
719
2.84M
  *rate = rd_cost_mbuv(x);
720
2.84M
  *distortion = vp8_mbuverror(x) / 4;
721
722
2.84M
  return RDCOST(x->rdmult, x->rddiv, *rate, *distortion);
723
2.84M
}
724
725
static int rd_inter4x4_uv(VP8_COMP *cpi, MACROBLOCK *x, int *rate,
726
386k
                          int *distortion, int fullpixel) {
727
386k
  (void)cpi;
728
386k
  (void)fullpixel;
729
730
386k
  vp8_build_inter4x4_predictors_mbuv(&x->e_mbd);
731
386k
  vp8_subtract_mbuv(x->src_diff, x->src.u_buffer, x->src.v_buffer,
732
386k
                    x->src.uv_stride, &x->e_mbd.predictor[256],
733
386k
                    &x->e_mbd.predictor[320], 8);
734
735
386k
  vp8_transform_mbuv(x);
736
386k
  vp8_quantize_mbuv(x);
737
738
386k
  *rate = rd_cost_mbuv(x);
739
386k
  *distortion = vp8_mbuverror(x) / 4;
740
741
386k
  return RDCOST(x->rdmult, x->rddiv, *rate, *distortion);
742
386k
}
743
744
static void rd_pick_intra_mbuv_mode(MACROBLOCK *x, int *rate,
745
1.29M
                                    int *rate_tokenonly, int *distortion) {
746
1.29M
  MB_PREDICTION_MODE mode;
747
1.29M
  MB_PREDICTION_MODE mode_selected = MB_MODE_COUNT;
748
1.29M
  int best_rd = INT_MAX;
749
1.29M
  int d = 0, r = 0;
750
1.29M
  int rate_to;
751
1.29M
  MACROBLOCKD *xd = &x->e_mbd;
752
753
6.46M
  for (mode = DC_PRED; mode <= TM_PRED; ++mode) {
754
5.17M
    int this_rate;
755
5.17M
    int this_distortion;
756
5.17M
    int this_rd;
757
758
5.17M
    xd->mode_info_context->mbmi.uv_mode = mode;
759
760
5.17M
    vp8_build_intra_predictors_mbuv_s(
761
5.17M
        xd, xd->dst.u_buffer - xd->dst.uv_stride,
762
5.17M
        xd->dst.v_buffer - xd->dst.uv_stride, xd->dst.u_buffer - 1,
763
5.17M
        xd->dst.v_buffer - 1, xd->dst.uv_stride, &xd->predictor[256],
764
5.17M
        &xd->predictor[320], 8);
765
766
5.17M
    vp8_subtract_mbuv(x->src_diff, x->src.u_buffer, x->src.v_buffer,
767
5.17M
                      x->src.uv_stride, &xd->predictor[256],
768
5.17M
                      &xd->predictor[320], 8);
769
5.17M
    vp8_transform_mbuv(x);
770
5.17M
    vp8_quantize_mbuv(x);
771
772
5.17M
    rate_to = rd_cost_mbuv(x);
773
5.17M
    this_rate =
774
5.17M
        rate_to + x->intra_uv_mode_cost[xd->frame_type]
775
5.17M
                                       [xd->mode_info_context->mbmi.uv_mode];
776
777
5.17M
    this_distortion = vp8_mbuverror(x) / 4;
778
779
5.17M
    this_rd = RDCOST(x->rdmult, x->rddiv, this_rate, this_distortion);
780
781
5.17M
    if (this_rd < best_rd) {
782
1.60M
      best_rd = this_rd;
783
1.60M
      d = this_distortion;
784
1.60M
      r = this_rate;
785
1.60M
      *rate_tokenonly = rate_to;
786
1.60M
      mode_selected = mode;
787
1.60M
    }
788
5.17M
  }
789
790
1.29M
  *rate = r;
791
1.29M
  *distortion = d;
792
793
1.29M
  assert(mode_selected != MB_MODE_COUNT);
794
1.29M
  xd->mode_info_context->mbmi.uv_mode = mode_selected;
795
1.29M
}
796
797
7.00M
int vp8_cost_mv_ref(MB_PREDICTION_MODE m, const int near_mv_ref_ct[4]) {
798
7.00M
  vp8_prob p[VP8_MVREFS - 1];
799
7.00M
  assert(NEARESTMV <= m && m <= SPLITMV);
800
7.00M
  vp8_mv_ref_probs(p, near_mv_ref_ct);
801
7.00M
  return vp8_cost_token(vp8_mv_ref_tree, p,
802
7.00M
                        vp8_mv_ref_encoding_array + (m - NEARESTMV));
803
7.00M
}
804
805
2.84M
void vp8_set_mbmode_and_mvs(MACROBLOCK *x, MB_PREDICTION_MODE mb, int_mv *mv) {
806
2.84M
  x->e_mbd.mode_info_context->mbmi.mode = mb;
807
2.84M
  x->e_mbd.mode_info_context->mbmi.mv.as_int = mv->as_int;
808
2.84M
}
809
810
static int labels2mode(MACROBLOCK *x, int const *labelings, int which_label,
811
                       B_PREDICTION_MODE this_mode, int_mv *this_mv,
812
30.0M
                       int_mv *best_ref_mv, int *mvcost[2]) {
813
30.0M
  MACROBLOCKD *const xd = &x->e_mbd;
814
30.0M
  MODE_INFO *const mic = xd->mode_info_context;
815
30.0M
  const int mis = xd->mode_info_stride;
816
817
30.0M
  int cost = 0;
818
30.0M
  int thismvcost = 0;
819
820
  /* We have to be careful retrieving previously-encoded motion vectors.
821
     Ones from this macroblock have to be pulled from the BLOCKD array
822
     as they have not yet made it to the bmi array in our MB_MODE_INFO. */
823
824
30.0M
  int i = 0;
825
826
480M
  do {
827
480M
    BLOCKD *const d = xd->block + i;
828
480M
    const int row = i >> 2, col = i & 3;
829
830
480M
    B_PREDICTION_MODE m;
831
832
480M
    if (labelings[i] != which_label) continue;
833
834
118M
    if (col && labelings[i] == labelings[i - 1]) {
835
61.3M
      m = LEFT4X4;
836
61.3M
    } else if (row && labelings[i] == labelings[i - 4]) {
837
27.0M
      m = ABOVE4X4;
838
30.0M
    } else {
839
      /* the only time we should do costing for new motion vector
840
       * or mode is when we are on a new label  (jbb May 08, 2007)
841
       */
842
30.0M
      switch (m = this_mode) {
843
8.21M
        case NEW4X4:
844
8.21M
          thismvcost = vp8_mv_bit_cost(this_mv, best_ref_mv, mvcost, 102);
845
8.21M
          break;
846
8.82M
        case LEFT4X4:
847
8.82M
          this_mv->as_int = col ? d[-1].bmi.mv.as_int : left_block_mv(mic, i);
848
8.82M
          break;
849
6.84M
        case ABOVE4X4:
850
6.84M
          this_mv->as_int =
851
6.84M
              row ? d[-4].bmi.mv.as_int : above_block_mv(mic, i, mis);
852
6.84M
          break;
853
6.17M
        case ZERO4X4: this_mv->as_int = 0; break;
854
0
        default: break;
855
30.0M
      }
856
857
30.0M
      if (m == ABOVE4X4) { /* replace above with left if same */
858
6.84M
        int_mv left_mv;
859
860
6.84M
        left_mv.as_int = col ? d[-1].bmi.mv.as_int : left_block_mv(mic, i);
861
862
6.84M
        if (left_mv.as_int == this_mv->as_int) m = LEFT4X4;
863
6.84M
      }
864
865
30.0M
      cost = x->inter_bmode_costs[m];
866
30.0M
    }
867
868
118M
    d->bmi.mv.as_int = this_mv->as_int;
869
870
118M
    x->partition_info->bmi[i].mode = m;
871
118M
    x->partition_info->bmi[i].mv.as_int = this_mv->as_int;
872
873
480M
  } while (++i < 16);
874
875
30.0M
  cost += thismvcost;
876
30.0M
  return cost;
877
30.0M
}
878
879
static int rdcost_mbsegment_y(MACROBLOCK *mb, const int *labels,
880
                              int which_label, ENTROPY_CONTEXT *ta,
881
23.6M
                              ENTROPY_CONTEXT *tl) {
882
23.6M
  int cost = 0;
883
23.6M
  int b;
884
23.6M
  MACROBLOCKD *x = &mb->e_mbd;
885
886
402M
  for (b = 0; b < 16; ++b) {
887
378M
    if (labels[b] == which_label) {
888
93.0M
      cost += cost_coeffs(mb, x->block + b, PLANE_TYPE_Y_WITH_DC,
889
93.0M
                          ta + vp8_block2above[b], tl + vp8_block2left[b]);
890
93.0M
    }
891
378M
  }
892
893
23.6M
  return cost;
894
23.6M
}
895
static unsigned int vp8_encode_inter_mb_segment(MACROBLOCK *x,
896
                                                int const *labels,
897
23.6M
                                                int which_label) {
898
23.6M
  int i;
899
23.6M
  unsigned int distortion = 0;
900
23.6M
  int pre_stride = x->e_mbd.pre.y_stride;
901
23.6M
  unsigned char *base_pre = x->e_mbd.pre.y_buffer;
902
903
402M
  for (i = 0; i < 16; ++i) {
904
378M
    if (labels[i] == which_label) {
905
93.0M
      BLOCKD *bd = &x->e_mbd.block[i];
906
93.0M
      BLOCK *be = &x->block[i];
907
908
93.0M
      vp8_build_inter_predictors_b(bd, 16, base_pre, pre_stride,
909
93.0M
                                   x->e_mbd.subpixel_predict);
910
93.0M
      vp8_subtract_b(be, bd, 16);
911
93.0M
      x->short_fdct4x4(be->src_diff, be->coeff, 32);
912
93.0M
      x->quantize_b(be, bd);
913
914
93.0M
      distortion += vp8_block_error(be->coeff, bd->dqcoeff);
915
93.0M
    }
916
378M
  }
917
918
23.6M
  return distortion;
919
23.6M
}
920
921
static const unsigned int segmentation_to_sseshift[4] = { 3, 3, 2, 0 };
922
923
typedef struct {
924
  int_mv *ref_mv;
925
  int_mv mvp;
926
927
  int segment_rd;
928
  int segment_num;
929
  int r;
930
  int d;
931
  int segment_yrate;
932
  B_PREDICTION_MODE modes[16];
933
  int_mv mvs[16];
934
  unsigned char eobs[16];
935
936
  int mvthresh;
937
  int *mdcounts;
938
939
  int_mv sv_mvp[4]; /* save 4 mvp from 8x8 */
940
  int sv_istep[2];  /* save 2 initial step_param for 16x8/8x16 */
941
942
} BEST_SEG_INFO;
943
944
static void rd_check_segment(VP8_COMP *cpi, MACROBLOCK *x, BEST_SEG_INFO *bsi,
945
1.64M
                             unsigned int segmentation) {
946
1.64M
  int i;
947
1.64M
  int const *labels;
948
1.64M
  int br = 0;
949
1.64M
  int bd = 0;
950
1.64M
  B_PREDICTION_MODE this_mode;
951
952
1.64M
  int label_count;
953
1.64M
  int this_segment_rd = 0;
954
1.64M
  int label_mv_thresh;
955
1.64M
  int rate = 0;
956
1.64M
  int sbr = 0;
957
1.64M
  int sbd = 0;
958
1.64M
  int segmentyrate = 0;
959
960
1.64M
  vp8_variance_fn_ptr_t *v_fn_ptr;
961
962
1.64M
  ENTROPY_CONTEXT_PLANES t_above, t_left;
963
1.64M
  ENTROPY_CONTEXT_PLANES t_above_b, t_left_b;
964
965
1.64M
  t_above = *x->e_mbd.above_context;
966
1.64M
  t_left = *x->e_mbd.left_context;
967
968
1.64M
  vp8_zero(t_above_b);
969
1.64M
  vp8_zero(t_left_b);
970
971
1.64M
  br = 0;
972
1.64M
  bd = 0;
973
974
1.64M
  v_fn_ptr = &cpi->fn_ptr[segmentation];
975
1.64M
  labels = vp8_mbsplits[segmentation];
976
1.64M
  label_count = vp8_mbsplit_count[segmentation];
977
978
  /* 64 makes this threshold really big effectively making it so that we
979
   * very rarely check mvs on segments.   setting this to 1 would make mv
980
   * thresh roughly equal to what it is for macroblocks
981
   */
982
1.64M
  label_mv_thresh = 1 * bsi->mvthresh / label_count;
983
984
  /* Segmentation method overheads */
985
1.64M
  rate = vp8_cost_token(vp8_mbsplit_tree, vp8_mbsplit_probs,
986
1.64M
                        vp8_mbsplit_encodings + segmentation);
987
1.64M
  rate += vp8_cost_mv_ref(SPLITMV, bsi->mdcounts);
988
1.64M
  this_segment_rd += RDCOST(x->rdmult, x->rddiv, rate, 0);
989
1.64M
  br += rate;
990
991
6.89M
  for (i = 0; i < label_count; ++i) {
992
6.10M
    int_mv mode_mv[B_MODE_COUNT] = { { 0 }, { 0 } };
993
6.10M
    int best_label_rd = INT_MAX;
994
6.10M
    B_PREDICTION_MODE mode_selected = ZERO4X4;
995
6.10M
    int bestlabelyrate = 0;
996
997
    /* search for the best motion vector on this segment */
998
30.0M
    for (this_mode = LEFT4X4; this_mode <= NEW4X4; ++this_mode) {
999
24.4M
      int this_rd;
1000
24.4M
      int distortion;
1001
24.4M
      int labelyrate;
1002
24.4M
      ENTROPY_CONTEXT_PLANES t_above_s, t_left_s;
1003
24.4M
      ENTROPY_CONTEXT *ta_s;
1004
24.4M
      ENTROPY_CONTEXT *tl_s;
1005
1006
24.4M
      t_above_s = t_above;
1007
24.4M
      t_left_s = t_left;
1008
1009
24.4M
      ta_s = (ENTROPY_CONTEXT *)&t_above_s;
1010
24.4M
      tl_s = (ENTROPY_CONTEXT *)&t_left_s;
1011
1012
24.4M
      if (this_mode == NEW4X4) {
1013
6.10M
        int sseshift;
1014
6.10M
        int num00;
1015
6.10M
        int step_param = 0;
1016
6.10M
        int further_steps;
1017
6.10M
        int n;
1018
6.10M
        int thissme;
1019
6.10M
        int bestsme = INT_MAX;
1020
6.10M
        int_mv temp_mv;
1021
6.10M
        BLOCK *c;
1022
6.10M
        BLOCKD *e;
1023
1024
        /* Is the best so far sufficiently good that we can't justify
1025
         * doing a new motion search.
1026
         */
1027
6.10M
        if (best_label_rd < label_mv_thresh) break;
1028
1029
5.63M
        if (cpi->compressor_speed) {
1030
5.63M
          if (segmentation == BLOCK_8X16 || segmentation == BLOCK_16X8) {
1031
1.43M
            bsi->mvp.as_int = bsi->sv_mvp[i].as_int;
1032
1.43M
            if (i == 1 && segmentation == BLOCK_16X8) {
1033
326k
              bsi->mvp.as_int = bsi->sv_mvp[2].as_int;
1034
326k
            }
1035
1036
1.43M
            step_param = bsi->sv_istep[i];
1037
1.43M
          }
1038
1039
          /* use previous block's result as next block's MV
1040
           * predictor.
1041
           */
1042
5.63M
          if (segmentation == BLOCK_4X4 && i > 0) {
1043
1.85M
            bsi->mvp.as_int = x->e_mbd.block[i - 1].bmi.mv.as_int;
1044
1.85M
            if (i == 4 || i == 8 || i == 12) {
1045
383k
              bsi->mvp.as_int = x->e_mbd.block[i - 4].bmi.mv.as_int;
1046
383k
            }
1047
1.85M
            step_param = 2;
1048
1.85M
          }
1049
5.63M
        }
1050
1051
5.63M
        further_steps = (MAX_MVSEARCH_STEPS - 1) - step_param;
1052
1053
5.63M
        {
1054
5.63M
          int sadpb = x->sadperbit4;
1055
5.63M
          int_mv mvp_full;
1056
1057
5.63M
          mvp_full.as_mv.row = bsi->mvp.as_mv.row >> 3;
1058
5.63M
          mvp_full.as_mv.col = bsi->mvp.as_mv.col >> 3;
1059
1060
          /* find first label */
1061
5.63M
          n = vp8_mbsplit_offset[segmentation][i];
1062
1063
5.63M
          c = &x->block[n];
1064
5.63M
          e = &x->e_mbd.block[n];
1065
1066
5.63M
          {
1067
5.63M
            bestsme = cpi->diamond_search_sad(
1068
5.63M
                x, c, e, &mvp_full, &mode_mv[NEW4X4], step_param, sadpb, &num00,
1069
5.63M
                v_fn_ptr, x->mvcost, bsi->ref_mv);
1070
1071
5.63M
            n = num00;
1072
5.63M
            num00 = 0;
1073
1074
21.8M
            while (n < further_steps) {
1075
16.2M
              n++;
1076
1077
16.2M
              if (num00) {
1078
2.01M
                num00--;
1079
14.2M
              } else {
1080
14.2M
                thissme = cpi->diamond_search_sad(
1081
14.2M
                    x, c, e, &mvp_full, &temp_mv, step_param + n, sadpb, &num00,
1082
14.2M
                    v_fn_ptr, x->mvcost, bsi->ref_mv);
1083
1084
14.2M
                if (thissme < bestsme) {
1085
2.69M
                  bestsme = thissme;
1086
2.69M
                  mode_mv[NEW4X4].as_int = temp_mv.as_int;
1087
2.69M
                }
1088
14.2M
              }
1089
16.2M
            }
1090
5.63M
          }
1091
1092
5.63M
          sseshift = segmentation_to_sseshift[segmentation];
1093
1094
          /* Should we do a full search (best quality only) */
1095
5.63M
          if ((cpi->compressor_speed == 0) && (bestsme >> sseshift) > 4000) {
1096
            /* Check if mvp_full is within the range. */
1097
0
            vp8_clamp_mv(&mvp_full, x->mv_col_min, x->mv_col_max, x->mv_row_min,
1098
0
                         x->mv_row_max);
1099
1100
0
            thissme = vp8_full_search_sad(x, c, e, &mvp_full, sadpb, 16,
1101
0
                                          v_fn_ptr, x->mvcost, bsi->ref_mv);
1102
1103
0
            if (thissme < bestsme) {
1104
0
              bestsme = thissme;
1105
0
              mode_mv[NEW4X4].as_int = e->bmi.mv.as_int;
1106
0
            } else {
1107
              /* The full search result is actually worse so
1108
               * re-instate the previous best vector
1109
               */
1110
0
              e->bmi.mv.as_int = mode_mv[NEW4X4].as_int;
1111
0
            }
1112
0
          }
1113
5.63M
        }
1114
1115
5.63M
        if (bestsme < INT_MAX) {
1116
5.63M
          int disto;
1117
5.63M
          unsigned int sse;
1118
5.63M
          cpi->find_fractional_mv_step(x, c, e, &mode_mv[NEW4X4], bsi->ref_mv,
1119
5.63M
                                       x->errorperbit, v_fn_ptr, x->mvcost,
1120
5.63M
                                       &disto, &sse);
1121
5.63M
        }
1122
5.63M
      } /* NEW4X4 */
1123
1124
23.9M
      rate = labels2mode(x, labels, i, this_mode, &mode_mv[this_mode],
1125
23.9M
                         bsi->ref_mv, x->mvcost);
1126
1127
      /* Trap vectors that reach beyond the UMV borders */
1128
23.9M
      if (((mode_mv[this_mode].as_mv.row >> 3) < x->mv_row_min) ||
1129
23.9M
          ((mode_mv[this_mode].as_mv.row >> 3) > x->mv_row_max) ||
1130
23.7M
          ((mode_mv[this_mode].as_mv.col >> 3) < x->mv_col_min) ||
1131
23.7M
          ((mode_mv[this_mode].as_mv.col >> 3) > x->mv_col_max)) {
1132
302k
        continue;
1133
302k
      }
1134
1135
23.6M
      distortion = vp8_encode_inter_mb_segment(x, labels, i) / 4;
1136
1137
23.6M
      labelyrate = rdcost_mbsegment_y(x, labels, i, ta_s, tl_s);
1138
23.6M
      rate += labelyrate;
1139
1140
23.6M
      this_rd = RDCOST(x->rdmult, x->rddiv, rate, distortion);
1141
1142
23.6M
      if (this_rd < best_label_rd) {
1143
10.3M
        sbr = rate;
1144
10.3M
        sbd = distortion;
1145
10.3M
        bestlabelyrate = labelyrate;
1146
10.3M
        mode_selected = this_mode;
1147
10.3M
        best_label_rd = this_rd;
1148
1149
10.3M
        t_above_b = t_above_s;
1150
10.3M
        t_left_b = t_left_s;
1151
10.3M
      }
1152
23.6M
    } /*for each 4x4 mode*/
1153
1154
6.10M
    t_above = t_above_b;
1155
6.10M
    t_left = t_left_b;
1156
1157
6.10M
    labels2mode(x, labels, i, mode_selected, &mode_mv[mode_selected],
1158
6.10M
                bsi->ref_mv, x->mvcost);
1159
1160
6.10M
    br += sbr;
1161
6.10M
    bd += sbd;
1162
6.10M
    segmentyrate += bestlabelyrate;
1163
6.10M
    this_segment_rd += best_label_rd;
1164
1165
6.10M
    if (this_segment_rd >= bsi->segment_rd) break;
1166
1167
6.10M
  } /* for each label */
1168
1169
1.64M
  if (this_segment_rd < bsi->segment_rd) {
1170
792k
    bsi->r = br;
1171
792k
    bsi->d = bd;
1172
792k
    bsi->segment_yrate = segmentyrate;
1173
792k
    bsi->segment_rd = this_segment_rd;
1174
792k
    bsi->segment_num = segmentation;
1175
1176
    /* store everything needed to come back to this!! */
1177
13.4M
    for (i = 0; i < 16; ++i) {
1178
12.6M
      bsi->mvs[i].as_mv = x->partition_info->bmi[i].mv.as_mv;
1179
12.6M
      bsi->modes[i] = x->partition_info->bmi[i].mode;
1180
12.6M
      bsi->eobs[i] = x->e_mbd.eobs[i];
1181
12.6M
    }
1182
792k
  }
1183
1.64M
}
1184
1185
1.54M
static void vp8_cal_step_param(int sr, int *sp) {
1186
1.54M
  int step = 0;
1187
1188
1.54M
  if (sr > MAX_FIRST_STEP) {
1189
43.2k
    sr = MAX_FIRST_STEP;
1190
1.50M
  } else if (sr < 1) {
1191
754k
    sr = 1;
1192
754k
  }
1193
1194
4.34M
  while (sr >>= 1) step++;
1195
1196
1.54M
  *sp = MAX_MVSEARCH_STEPS - 1 - step;
1197
1.54M
}
1198
1199
static int vp8_rd_pick_best_mbsegmentation(VP8_COMP *cpi, MACROBLOCK *x,
1200
                                           int_mv *best_ref_mv, int best_rd,
1201
                                           int *mdcounts, int *returntotrate,
1202
                                           int *returnyrate,
1203
                                           int *returndistortion,
1204
730k
                                           int mvthresh) {
1205
730k
  int i;
1206
730k
  BEST_SEG_INFO bsi;
1207
1208
730k
  memset(&bsi, 0, sizeof(bsi));
1209
1210
730k
  bsi.segment_rd = best_rd;
1211
730k
  bsi.ref_mv = best_ref_mv;
1212
730k
  bsi.mvp.as_int = best_ref_mv->as_int;
1213
730k
  bsi.mvthresh = mvthresh;
1214
730k
  bsi.mdcounts = mdcounts;
1215
1216
12.4M
  for (i = 0; i < 16; ++i) {
1217
11.6M
    bsi.modes[i] = ZERO4X4;
1218
11.6M
  }
1219
1220
730k
  if (cpi->compressor_speed == 0) {
1221
    /* for now, we will keep the original segmentation order
1222
       when in best quality mode */
1223
0
    rd_check_segment(cpi, x, &bsi, BLOCK_16X8);
1224
0
    rd_check_segment(cpi, x, &bsi, BLOCK_8X16);
1225
0
    rd_check_segment(cpi, x, &bsi, BLOCK_8X8);
1226
0
    rd_check_segment(cpi, x, &bsi, BLOCK_4X4);
1227
730k
  } else {
1228
730k
    int sr;
1229
1230
730k
    rd_check_segment(cpi, x, &bsi, BLOCK_8X8);
1231
1232
730k
    if (bsi.segment_rd < best_rd) {
1233
386k
      int col_min = ((best_ref_mv->as_mv.col + 7) >> 3) - MAX_FULL_PEL_VAL;
1234
386k
      int row_min = ((best_ref_mv->as_mv.row + 7) >> 3) - MAX_FULL_PEL_VAL;
1235
386k
      int col_max = (best_ref_mv->as_mv.col >> 3) + MAX_FULL_PEL_VAL;
1236
386k
      int row_max = (best_ref_mv->as_mv.row >> 3) + MAX_FULL_PEL_VAL;
1237
1238
386k
      int tmp_col_min = x->mv_col_min;
1239
386k
      int tmp_col_max = x->mv_col_max;
1240
386k
      int tmp_row_min = x->mv_row_min;
1241
386k
      int tmp_row_max = x->mv_row_max;
1242
1243
      /* Get intersection of UMV window and valid MV window to reduce # of
1244
       * checks in diamond search. */
1245
386k
      if (x->mv_col_min < col_min) x->mv_col_min = col_min;
1246
386k
      if (x->mv_col_max > col_max) x->mv_col_max = col_max;
1247
386k
      if (x->mv_row_min < row_min) x->mv_row_min = row_min;
1248
386k
      if (x->mv_row_max > row_max) x->mv_row_max = row_max;
1249
1250
      /* Get 8x8 result */
1251
386k
      bsi.sv_mvp[0].as_int = bsi.mvs[0].as_int;
1252
386k
      bsi.sv_mvp[1].as_int = bsi.mvs[2].as_int;
1253
386k
      bsi.sv_mvp[2].as_int = bsi.mvs[8].as_int;
1254
386k
      bsi.sv_mvp[3].as_int = bsi.mvs[10].as_int;
1255
1256
      /* Use 8x8 result as 16x8/8x16's predictor MV. Adjust search range
1257
       * according to the closeness of 2 MV. */
1258
      /* block 8X16 */
1259
386k
      {
1260
386k
        sr =
1261
386k
            MAXF((abs(bsi.sv_mvp[0].as_mv.row - bsi.sv_mvp[2].as_mv.row)) >> 3,
1262
386k
                 (abs(bsi.sv_mvp[0].as_mv.col - bsi.sv_mvp[2].as_mv.col)) >> 3);
1263
386k
        vp8_cal_step_param(sr, &bsi.sv_istep[0]);
1264
1265
386k
        sr =
1266
386k
            MAXF((abs(bsi.sv_mvp[1].as_mv.row - bsi.sv_mvp[3].as_mv.row)) >> 3,
1267
386k
                 (abs(bsi.sv_mvp[1].as_mv.col - bsi.sv_mvp[3].as_mv.col)) >> 3);
1268
386k
        vp8_cal_step_param(sr, &bsi.sv_istep[1]);
1269
1270
386k
        rd_check_segment(cpi, x, &bsi, BLOCK_8X16);
1271
386k
      }
1272
1273
      /* block 16X8 */
1274
386k
      {
1275
386k
        sr =
1276
386k
            MAXF((abs(bsi.sv_mvp[0].as_mv.row - bsi.sv_mvp[1].as_mv.row)) >> 3,
1277
386k
                 (abs(bsi.sv_mvp[0].as_mv.col - bsi.sv_mvp[1].as_mv.col)) >> 3);
1278
386k
        vp8_cal_step_param(sr, &bsi.sv_istep[0]);
1279
1280
386k
        sr =
1281
386k
            MAXF((abs(bsi.sv_mvp[2].as_mv.row - bsi.sv_mvp[3].as_mv.row)) >> 3,
1282
386k
                 (abs(bsi.sv_mvp[2].as_mv.col - bsi.sv_mvp[3].as_mv.col)) >> 3);
1283
386k
        vp8_cal_step_param(sr, &bsi.sv_istep[1]);
1284
1285
386k
        rd_check_segment(cpi, x, &bsi, BLOCK_16X8);
1286
386k
      }
1287
1288
      /* If 8x8 is better than 16x8/8x16, then do 4x4 search */
1289
      /* Not skip 4x4 if speed=0 (good quality) */
1290
386k
      if (cpi->sf.no_skip_block4x4_search || bsi.segment_num == BLOCK_8X8)
1291
      /* || (sv_segment_rd8x8-bsi.segment_rd) < sv_segment_rd8x8>>5) */
1292
139k
      {
1293
139k
        bsi.mvp.as_int = bsi.sv_mvp[0].as_int;
1294
139k
        rd_check_segment(cpi, x, &bsi, BLOCK_4X4);
1295
139k
      }
1296
1297
      /* restore UMV window */
1298
386k
      x->mv_col_min = tmp_col_min;
1299
386k
      x->mv_col_max = tmp_col_max;
1300
386k
      x->mv_row_min = tmp_row_min;
1301
386k
      x->mv_row_max = tmp_row_max;
1302
386k
    }
1303
730k
  }
1304
1305
  /* set it to the best */
1306
12.4M
  for (i = 0; i < 16; ++i) {
1307
11.6M
    BLOCKD *bd = &x->e_mbd.block[i];
1308
1309
11.6M
    bd->bmi.mv.as_int = bsi.mvs[i].as_int;
1310
11.6M
    *bd->eob = bsi.eobs[i];
1311
11.6M
  }
1312
1313
730k
  *returntotrate = bsi.r;
1314
730k
  *returndistortion = bsi.d;
1315
730k
  *returnyrate = bsi.segment_yrate;
1316
1317
  /* save partitions */
1318
730k
  x->e_mbd.mode_info_context->mbmi.partitioning = bsi.segment_num;
1319
730k
  x->partition_info->count = vp8_mbsplit_count[bsi.segment_num];
1320
1321
3.59M
  for (i = 0; i < x->partition_info->count; ++i) {
1322
2.86M
    int j;
1323
1324
2.86M
    j = vp8_mbsplit_offset[bsi.segment_num][i];
1325
1326
2.86M
    x->partition_info->bmi[i].mode = bsi.modes[j];
1327
2.86M
    x->partition_info->bmi[i].mv.as_mv = bsi.mvs[j].as_mv;
1328
2.86M
  }
1329
  /*
1330
   * used to set x->e_mbd.mode_info_context->mbmi.mv.as_int
1331
   */
1332
730k
  x->partition_info->bmi[15].mv.as_int = bsi.mvs[15].as_int;
1333
1334
730k
  return bsi.segment_rd;
1335
730k
}
1336
1337
/* The improved MV prediction */
1338
void vp8_mv_pred(VP8_COMP *cpi, MACROBLOCKD *xd, const MODE_INFO *here,
1339
                 int_mv *mvp, int refframe, int *ref_frame_sign_bias, int *sr,
1340
1.60M
                 int near_sadidx[]) {
1341
1.60M
  const MODE_INFO *above = here - xd->mode_info_stride;
1342
1.60M
  const MODE_INFO *left = here - 1;
1343
1.60M
  const MODE_INFO *aboveleft = above - 1;
1344
1.60M
  int_mv near_mvs[8];
1345
1.60M
  int near_ref[8];
1346
1.60M
  int_mv mv;
1347
1.60M
  int vcnt = 0;
1348
1.60M
  int find = 0;
1349
1.60M
  int mb_offset;
1350
1351
1.60M
  int mvx[8];
1352
1.60M
  int mvy[8];
1353
1.60M
  int i;
1354
1355
1.60M
  mv.as_int = 0;
1356
1357
1.60M
  if (here->mbmi.ref_frame != INTRA_FRAME) {
1358
1.60M
    near_mvs[0].as_int = near_mvs[1].as_int = near_mvs[2].as_int =
1359
1.60M
        near_mvs[3].as_int = near_mvs[4].as_int = near_mvs[5].as_int =
1360
1.60M
            near_mvs[6].as_int = near_mvs[7].as_int = 0;
1361
1.60M
    near_ref[0] = near_ref[1] = near_ref[2] = near_ref[3] = near_ref[4] =
1362
1.60M
        near_ref[5] = near_ref[6] = near_ref[7] = 0;
1363
1364
    /* read in 3 nearby block's MVs from current frame as prediction
1365
     * candidates.
1366
     */
1367
1.60M
    if (above->mbmi.ref_frame != INTRA_FRAME) {
1368
509k
      near_mvs[vcnt].as_int = above->mbmi.mv.as_int;
1369
509k
      mv_bias(ref_frame_sign_bias[above->mbmi.ref_frame], refframe,
1370
509k
              &near_mvs[vcnt], ref_frame_sign_bias);
1371
509k
      near_ref[vcnt] = above->mbmi.ref_frame;
1372
509k
    }
1373
1.60M
    vcnt++;
1374
1.60M
    if (left->mbmi.ref_frame != INTRA_FRAME) {
1375
666k
      near_mvs[vcnt].as_int = left->mbmi.mv.as_int;
1376
666k
      mv_bias(ref_frame_sign_bias[left->mbmi.ref_frame], refframe,
1377
666k
              &near_mvs[vcnt], ref_frame_sign_bias);
1378
666k
      near_ref[vcnt] = left->mbmi.ref_frame;
1379
666k
    }
1380
1.60M
    vcnt++;
1381
1.60M
    if (aboveleft->mbmi.ref_frame != INTRA_FRAME) {
1382
389k
      near_mvs[vcnt].as_int = aboveleft->mbmi.mv.as_int;
1383
389k
      mv_bias(ref_frame_sign_bias[aboveleft->mbmi.ref_frame], refframe,
1384
389k
              &near_mvs[vcnt], ref_frame_sign_bias);
1385
389k
      near_ref[vcnt] = aboveleft->mbmi.ref_frame;
1386
389k
    }
1387
1.60M
    vcnt++;
1388
1389
    /* read in 5 nearby block's MVs from last frame. */
1390
1.60M
    if (cpi->common.last_frame_type != KEY_FRAME) {
1391
1.00M
      mb_offset = (-xd->mb_to_top_edge / 128 + 1) * (xd->mode_info_stride + 1) +
1392
1.00M
                  (-xd->mb_to_left_edge / 128 + 1);
1393
1394
      /* current in last frame */
1395
1.00M
      if (cpi->lf_ref_frame[mb_offset] != INTRA_FRAME) {
1396
621k
        near_mvs[vcnt].as_int = cpi->lfmv[mb_offset].as_int;
1397
621k
        mv_bias(cpi->lf_ref_frame_sign_bias[mb_offset], refframe,
1398
621k
                &near_mvs[vcnt], ref_frame_sign_bias);
1399
621k
        near_ref[vcnt] = cpi->lf_ref_frame[mb_offset];
1400
621k
      }
1401
1.00M
      vcnt++;
1402
1403
      /* above in last frame */
1404
1.00M
      if (cpi->lf_ref_frame[mb_offset - xd->mode_info_stride - 1] !=
1405
1.00M
          INTRA_FRAME) {
1406
353k
        near_mvs[vcnt].as_int =
1407
353k
            cpi->lfmv[mb_offset - xd->mode_info_stride - 1].as_int;
1408
353k
        mv_bias(
1409
353k
            cpi->lf_ref_frame_sign_bias[mb_offset - xd->mode_info_stride - 1],
1410
353k
            refframe, &near_mvs[vcnt], ref_frame_sign_bias);
1411
353k
        near_ref[vcnt] =
1412
353k
            cpi->lf_ref_frame[mb_offset - xd->mode_info_stride - 1];
1413
353k
      }
1414
1.00M
      vcnt++;
1415
1416
      /* left in last frame */
1417
1.00M
      if (cpi->lf_ref_frame[mb_offset - 1] != INTRA_FRAME) {
1418
410k
        near_mvs[vcnt].as_int = cpi->lfmv[mb_offset - 1].as_int;
1419
410k
        mv_bias(cpi->lf_ref_frame_sign_bias[mb_offset - 1], refframe,
1420
410k
                &near_mvs[vcnt], ref_frame_sign_bias);
1421
410k
        near_ref[vcnt] = cpi->lf_ref_frame[mb_offset - 1];
1422
410k
      }
1423
1.00M
      vcnt++;
1424
1425
      /* right in last frame */
1426
1.00M
      if (cpi->lf_ref_frame[mb_offset + 1] != INTRA_FRAME) {
1427
415k
        near_mvs[vcnt].as_int = cpi->lfmv[mb_offset + 1].as_int;
1428
415k
        mv_bias(cpi->lf_ref_frame_sign_bias[mb_offset + 1], refframe,
1429
415k
                &near_mvs[vcnt], ref_frame_sign_bias);
1430
415k
        near_ref[vcnt] = cpi->lf_ref_frame[mb_offset + 1];
1431
415k
      }
1432
1.00M
      vcnt++;
1433
1434
      /* below in last frame */
1435
1.00M
      if (cpi->lf_ref_frame[mb_offset + xd->mode_info_stride + 1] !=
1436
1.00M
          INTRA_FRAME) {
1437
344k
        near_mvs[vcnt].as_int =
1438
344k
            cpi->lfmv[mb_offset + xd->mode_info_stride + 1].as_int;
1439
344k
        mv_bias(
1440
344k
            cpi->lf_ref_frame_sign_bias[mb_offset + xd->mode_info_stride + 1],
1441
344k
            refframe, &near_mvs[vcnt], ref_frame_sign_bias);
1442
344k
        near_ref[vcnt] =
1443
344k
            cpi->lf_ref_frame[mb_offset + xd->mode_info_stride + 1];
1444
344k
      }
1445
1.00M
      vcnt++;
1446
1.00M
    }
1447
1448
6.45M
    for (i = 0; i < vcnt; ++i) {
1449
5.77M
      if (near_ref[near_sadidx[i]] != INTRA_FRAME) {
1450
2.06M
        if (here->mbmi.ref_frame == near_ref[near_sadidx[i]]) {
1451
923k
          mv.as_int = near_mvs[near_sadidx[i]].as_int;
1452
923k
          find = 1;
1453
923k
          if (i < 3) {
1454
836k
            *sr = 3;
1455
836k
          } else {
1456
87.1k
            *sr = 2;
1457
87.1k
          }
1458
923k
          break;
1459
923k
        }
1460
2.06M
      }
1461
5.77M
    }
1462
1463
1.60M
    if (!find) {
1464
4.84M
      for (i = 0; i < vcnt; ++i) {
1465
4.16M
        mvx[i] = near_mvs[i].as_mv.row;
1466
4.16M
        mvy[i] = near_mvs[i].as_mv.col;
1467
4.16M
      }
1468
1469
680k
      insertsortmv(mvx, vcnt);
1470
680k
      insertsortmv(mvy, vcnt);
1471
680k
      mv.as_mv.row = mvx[vcnt / 2];
1472
680k
      mv.as_mv.col = mvy[vcnt / 2];
1473
1474
      /* sr is set to 0 to allow calling function to decide the search
1475
       * range.
1476
       */
1477
680k
      *sr = 0;
1478
680k
    }
1479
1.60M
  }
1480
1481
  /* Set up return values */
1482
1.60M
  mvp->as_int = mv.as_int;
1483
1.60M
  vp8_clamp_mv2(mvp, xd);
1484
1.60M
}
1485
1486
void vp8_cal_sad(VP8_COMP *cpi, MACROBLOCKD *xd, MACROBLOCK *x,
1487
1.15M
                 int recon_yoffset, int near_sadidx[]) {
1488
  /* near_sad indexes:
1489
   *   0-cf above, 1-cf left, 2-cf aboveleft,
1490
   *   3-lf current, 4-lf above, 5-lf left, 6-lf right, 7-lf below
1491
   */
1492
1.15M
  int near_sad[8] = { 0 };
1493
1.15M
  BLOCK *b = &x->block[0];
1494
1.15M
  unsigned char *src_y_ptr = *(b->base_src);
1495
1496
  /* calculate sad for current frame 3 nearby MBs. */
1497
1.15M
  if (xd->mb_to_top_edge == 0 && xd->mb_to_left_edge == 0) {
1498
89.9k
    near_sad[0] = near_sad[1] = near_sad[2] = INT_MAX;
1499
1.06M
  } else if (xd->mb_to_top_edge ==
1500
1.06M
             0) { /* only has left MB for sad calculation. */
1501
399k
    near_sad[0] = near_sad[2] = INT_MAX;
1502
399k
    near_sad[1] = cpi->fn_ptr[BLOCK_16X16].sdf(
1503
399k
        src_y_ptr, b->src_stride, xd->dst.y_buffer - 16, xd->dst.y_stride);
1504
664k
  } else if (xd->mb_to_left_edge ==
1505
664k
             0) { /* only has left MB for sad calculation. */
1506
121k
    near_sad[1] = near_sad[2] = INT_MAX;
1507
121k
    near_sad[0] = cpi->fn_ptr[BLOCK_16X16].sdf(
1508
121k
        src_y_ptr, b->src_stride, xd->dst.y_buffer - xd->dst.y_stride * 16,
1509
121k
        xd->dst.y_stride);
1510
542k
  } else {
1511
542k
    near_sad[0] = cpi->fn_ptr[BLOCK_16X16].sdf(
1512
542k
        src_y_ptr, b->src_stride, xd->dst.y_buffer - xd->dst.y_stride * 16,
1513
542k
        xd->dst.y_stride);
1514
542k
    near_sad[1] = cpi->fn_ptr[BLOCK_16X16].sdf(
1515
542k
        src_y_ptr, b->src_stride, xd->dst.y_buffer - 16, xd->dst.y_stride);
1516
542k
    near_sad[2] = cpi->fn_ptr[BLOCK_16X16].sdf(
1517
542k
        src_y_ptr, b->src_stride, xd->dst.y_buffer - xd->dst.y_stride * 16 - 16,
1518
542k
        xd->dst.y_stride);
1519
542k
  }
1520
1521
1.15M
  if (cpi->common.last_frame_type != KEY_FRAME) {
1522
    /* calculate sad for last frame 5 nearby MBs. */
1523
550k
    unsigned char *pre_y_buffer =
1524
550k
        cpi->common.yv12_fb[cpi->common.lst_fb_idx].y_buffer + recon_yoffset;
1525
550k
    int pre_y_stride = cpi->common.yv12_fb[cpi->common.lst_fb_idx].y_stride;
1526
1527
550k
    if (xd->mb_to_top_edge == 0) near_sad[4] = INT_MAX;
1528
550k
    if (xd->mb_to_left_edge == 0) near_sad[5] = INT_MAX;
1529
550k
    if (xd->mb_to_right_edge == 0) near_sad[6] = INT_MAX;
1530
550k
    if (xd->mb_to_bottom_edge == 0) near_sad[7] = INT_MAX;
1531
1532
550k
    if (near_sad[4] != INT_MAX) {
1533
334k
      near_sad[4] = cpi->fn_ptr[BLOCK_16X16].sdf(
1534
334k
          src_y_ptr, b->src_stride, pre_y_buffer - pre_y_stride * 16,
1535
334k
          pre_y_stride);
1536
334k
    }
1537
550k
    if (near_sad[5] != INT_MAX) {
1538
402k
      near_sad[5] = cpi->fn_ptr[BLOCK_16X16].sdf(
1539
402k
          src_y_ptr, b->src_stride, pre_y_buffer - 16, pre_y_stride);
1540
402k
    }
1541
550k
    near_sad[3] = cpi->fn_ptr[BLOCK_16X16].sdf(src_y_ptr, b->src_stride,
1542
550k
                                               pre_y_buffer, pre_y_stride);
1543
550k
    if (near_sad[6] != INT_MAX) {
1544
404k
      near_sad[6] = cpi->fn_ptr[BLOCK_16X16].sdf(
1545
404k
          src_y_ptr, b->src_stride, pre_y_buffer + 16, pre_y_stride);
1546
404k
    }
1547
550k
    if (near_sad[7] != INT_MAX) {
1548
354k
      near_sad[7] = cpi->fn_ptr[BLOCK_16X16].sdf(
1549
354k
          src_y_ptr, b->src_stride, pre_y_buffer + pre_y_stride * 16,
1550
354k
          pre_y_stride);
1551
354k
    }
1552
550k
  }
1553
1554
1.15M
  if (cpi->common.last_frame_type != KEY_FRAME) {
1555
550k
    insertsortsad(near_sad, near_sadidx, 8);
1556
603k
  } else {
1557
603k
    insertsortsad(near_sad, near_sadidx, 3);
1558
603k
  }
1559
1.15M
}
1560
1561
771k
static void rd_update_mvcount(MACROBLOCK *x, int_mv *best_ref_mv) {
1562
771k
  if (x->e_mbd.mode_info_context->mbmi.mode == SPLITMV) {
1563
173k
    int i;
1564
1565
1.43M
    for (i = 0; i < x->partition_info->count; ++i) {
1566
1.25M
      if (x->partition_info->bmi[i].mode == NEW4X4) {
1567
536k
        const int row_val = ((x->partition_info->bmi[i].mv.as_mv.row -
1568
536k
                              best_ref_mv->as_mv.row) >>
1569
536k
                             1);
1570
536k
        const int row_idx = mv_max + row_val;
1571
536k
        const int col_val = ((x->partition_info->bmi[i].mv.as_mv.col -
1572
536k
                              best_ref_mv->as_mv.col) >>
1573
536k
                             1);
1574
536k
        const int col_idx = mv_max + col_val;
1575
536k
        if (row_idx >= 0 && row_idx < MVvals && col_idx >= 0 &&
1576
536k
            col_idx < MVvals) {
1577
536k
          x->MVcount[0][row_idx]++;
1578
536k
          x->MVcount[1][col_idx]++;
1579
536k
        }
1580
536k
      }
1581
1.25M
    }
1582
598k
  } else if (x->e_mbd.mode_info_context->mbmi.mode == NEWMV) {
1583
102k
    const int row_val = ((x->e_mbd.mode_info_context->mbmi.mv.as_mv.row -
1584
102k
                          best_ref_mv->as_mv.row) >>
1585
102k
                         1);
1586
102k
    const int row_idx = mv_max + row_val;
1587
102k
    const int col_val = ((x->e_mbd.mode_info_context->mbmi.mv.as_mv.col -
1588
102k
                          best_ref_mv->as_mv.col) >>
1589
102k
                         1);
1590
102k
    const int col_idx = mv_max + col_val;
1591
102k
    if (row_idx >= 0 && row_idx < MVvals && col_idx >= 0 && col_idx < MVvals) {
1592
102k
      x->MVcount[0][row_idx]++;
1593
102k
      x->MVcount[1][col_idx]++;
1594
102k
    }
1595
102k
  }
1596
771k
}
1597
1598
static int evaluate_inter_mode_rd(int mdcounts[4], RATE_DISTORTION *rd,
1599
                                  int *disable_skip, VP8_COMP *cpi,
1600
2.84M
                                  MACROBLOCK *x) {
1601
2.84M
  MB_PREDICTION_MODE this_mode = x->e_mbd.mode_info_context->mbmi.mode;
1602
2.84M
  BLOCK *b = &x->block[0];
1603
2.84M
  MACROBLOCKD *xd = &x->e_mbd;
1604
2.84M
  int distortion;
1605
2.84M
  vp8_build_inter16x16_predictors_mby(&x->e_mbd, x->e_mbd.predictor, 16);
1606
1607
2.84M
  if (cpi->active_map_enabled && x->active_ptr[0] == 0) {
1608
0
    x->skip = 1;
1609
2.84M
  } else if (x->encode_breakout) {
1610
0
    unsigned int sse;
1611
0
    unsigned int var;
1612
0
    unsigned int threshold =
1613
0
        (xd->block[0].dequant[1] * xd->block[0].dequant[1] >> 4);
1614
1615
0
    if (threshold < x->encode_breakout) threshold = x->encode_breakout;
1616
1617
0
    var = vpx_variance16x16(*(b->base_src), b->src_stride, x->e_mbd.predictor,
1618
0
                            16, &sse);
1619
1620
0
    if (sse < threshold) {
1621
0
      unsigned int q2dc = xd->block[24].dequant[0];
1622
      /* If theres is no codeable 2nd order dc
1623
         or a very small uniform pixel change change */
1624
0
      if ((sse - var < q2dc * q2dc >> 4) || (sse / 2 > var && sse - var < 64)) {
1625
        /* Check u and v to make sure skip is ok */
1626
0
        unsigned int sse2 = VP8_UVSSE(x);
1627
0
        if (sse2 * 2 < threshold) {
1628
0
          x->skip = 1;
1629
0
          rd->distortion2 = sse + sse2;
1630
0
          rd->rate2 = 500;
1631
1632
          /* for best_yrd calculation */
1633
0
          rd->rate_uv = 0;
1634
0
          rd->distortion_uv = sse2;
1635
1636
0
          *disable_skip = 1;
1637
0
          return RDCOST(x->rdmult, x->rddiv, rd->rate2, rd->distortion2);
1638
0
        }
1639
0
      }
1640
0
    }
1641
0
  }
1642
1643
  /* Add in the Mv/mode cost */
1644
2.84M
  rd->rate2 += vp8_cost_mv_ref(this_mode, mdcounts);
1645
1646
  /* Y cost and distortion */
1647
2.84M
  macro_block_yrd(x, &rd->rate_y, &distortion);
1648
2.84M
  rd->rate2 += rd->rate_y;
1649
2.84M
  rd->distortion2 += distortion;
1650
1651
  /* UV cost and distortion */
1652
2.84M
  rd_inter16x16_uv(cpi, x, &rd->rate_uv, &rd->distortion_uv,
1653
2.84M
                   cpi->common.full_pixel);
1654
2.84M
  rd->rate2 += rd->rate_uv;
1655
2.84M
  rd->distortion2 += rd->distortion_uv;
1656
2.84M
  return INT_MAX;
1657
2.84M
}
1658
1659
static int calculate_final_rd_costs(int this_rd, RATE_DISTORTION *rd,
1660
                                    int *other_cost, int disable_skip,
1661
                                    int uv_intra_tteob, int intra_rd_penalty,
1662
6.76M
                                    VP8_COMP *cpi, MACROBLOCK *x) {
1663
6.76M
  MB_PREDICTION_MODE this_mode = x->e_mbd.mode_info_context->mbmi.mode;
1664
1665
  /* Where skip is allowable add in the default per mb cost for the no
1666
   * skip case. where we then decide to skip we have to delete this and
1667
   * replace it with the cost of signalling a skip
1668
   */
1669
6.76M
  if (cpi->common.mb_no_coeff_skip) {
1670
6.76M
    *other_cost += vp8_cost_bit(cpi->prob_skip_false, 0);
1671
6.76M
    rd->rate2 += *other_cost;
1672
6.76M
  }
1673
1674
  /* Estimate the reference frame signaling cost and add it
1675
   * to the rolling cost variable.
1676
   */
1677
6.76M
  rd->rate2 += x->ref_frame_cost[x->e_mbd.mode_info_context->mbmi.ref_frame];
1678
1679
6.76M
  if (!disable_skip) {
1680
    /* Test for the condition where skip block will be activated
1681
     * because there are no non zero coefficients and make any
1682
     * necessary adjustment for rate
1683
     */
1684
6.09M
    if (cpi->common.mb_no_coeff_skip) {
1685
6.09M
      int i;
1686
6.09M
      int tteob;
1687
6.09M
      int has_y2_block = (this_mode != SPLITMV && this_mode != B_PRED);
1688
1689
6.09M
      tteob = 0;
1690
6.09M
      if (has_y2_block) tteob += x->e_mbd.eobs[24];
1691
1692
103M
      for (i = 0; i < 16; ++i) tteob += (x->e_mbd.eobs[i] > has_y2_block);
1693
1694
6.09M
      if (x->e_mbd.mode_info_context->mbmi.ref_frame) {
1695
29.1M
        for (i = 16; i < 24; ++i) tteob += x->e_mbd.eobs[i];
1696
3.23M
      } else {
1697
2.86M
        tteob += uv_intra_tteob;
1698
2.86M
      }
1699
1700
6.09M
      if (tteob == 0) {
1701
395k
        rd->rate2 -= (rd->rate_y + rd->rate_uv);
1702
        /* for best_yrd calculation */
1703
395k
        rd->rate_uv = 0;
1704
1705
        /* Back out no skip flag costing and add in skip flag costing */
1706
395k
        if (cpi->prob_skip_false) {
1707
395k
          int prob_skip_cost;
1708
1709
395k
          prob_skip_cost = vp8_cost_bit(cpi->prob_skip_false, 1);
1710
395k
          prob_skip_cost -= (int)vp8_cost_bit(cpi->prob_skip_false, 0);
1711
395k
          rd->rate2 += prob_skip_cost;
1712
395k
          *other_cost += prob_skip_cost;
1713
395k
        }
1714
395k
      }
1715
6.09M
    }
1716
    /* Calculate the final RD estimate for this mode */
1717
6.09M
    this_rd = RDCOST(x->rdmult, x->rddiv, rd->rate2, rd->distortion2);
1718
6.09M
    if (this_rd < INT_MAX &&
1719
6.09M
        x->e_mbd.mode_info_context->mbmi.ref_frame == INTRA_FRAME) {
1720
2.86M
      this_rd += intra_rd_penalty;
1721
2.86M
    }
1722
6.09M
  }
1723
6.76M
  return this_rd;
1724
6.76M
}
1725
1726
static void update_best_mode(BEST_MODE *best_mode, int this_rd,
1727
                             RATE_DISTORTION *rd, int other_cost,
1728
2.51M
                             MACROBLOCK *x) {
1729
2.51M
  MB_PREDICTION_MODE this_mode = x->e_mbd.mode_info_context->mbmi.mode;
1730
1731
2.51M
  other_cost += x->ref_frame_cost[x->e_mbd.mode_info_context->mbmi.ref_frame];
1732
1733
  /* Calculate the final y RD estimate for this mode */
1734
2.51M
  best_mode->yrd =
1735
2.51M
      RDCOST(x->rdmult, x->rddiv, (rd->rate2 - rd->rate_uv - other_cost),
1736
2.51M
             (rd->distortion2 - rd->distortion_uv));
1737
1738
2.51M
  best_mode->rd = this_rd;
1739
2.51M
  best_mode->mbmode = x->e_mbd.mode_info_context->mbmi;
1740
2.51M
  best_mode->partition = *x->partition_info;
1741
1742
2.51M
  if ((this_mode == B_PRED) || (this_mode == SPLITMV)) {
1743
525k
    int i;
1744
8.93M
    for (i = 0; i < 16; ++i) {
1745
8.40M
      best_mode->bmodes[i] = x->e_mbd.block[i].bmi;
1746
8.40M
    }
1747
525k
  }
1748
2.51M
}
1749
1750
void vp8_rd_pick_inter_mode(VP8_COMP *cpi, MACROBLOCK *x, int recon_yoffset,
1751
                            int recon_uvoffset, int *returnrate,
1752
                            int *returndistortion, int *returnintra, int mb_row,
1753
771k
                            int mb_col) {
1754
771k
  BLOCK *b = &x->block[0];
1755
771k
  BLOCKD *d = &x->e_mbd.block[0];
1756
771k
  MACROBLOCKD *xd = &x->e_mbd;
1757
771k
  int_mv best_ref_mv_sb[2];
1758
771k
  int_mv mode_mv_sb[2][MB_MODE_COUNT];
1759
771k
  int_mv best_ref_mv;
1760
771k
  int_mv *mode_mv;
1761
771k
  MB_PREDICTION_MODE this_mode;
1762
771k
  int num00;
1763
771k
  int best_mode_index = 0;
1764
771k
  BEST_MODE best_mode;
1765
1766
771k
  int i;
1767
771k
  int mode_index;
1768
771k
  int mdcounts[4];
1769
771k
  int rate;
1770
771k
  RATE_DISTORTION rd;
1771
771k
  int uv_intra_rate, uv_intra_distortion, uv_intra_rate_tokenonly;
1772
771k
  int uv_intra_tteob = 0;
1773
771k
  int uv_intra_done = 0;
1774
1775
771k
  MB_PREDICTION_MODE uv_intra_mode = 0;
1776
771k
  int_mv mvp;
1777
771k
  int near_sadidx[8] = { 0, 1, 2, 3, 4, 5, 6, 7 };
1778
771k
  int saddone = 0;
1779
  /* search range got from mv_pred(). It uses step_param levels. (0-7) */
1780
771k
  int sr = 0;
1781
1782
771k
  unsigned char *plane[4][3] = { { 0, 0 } };
1783
771k
  int ref_frame_map[4];
1784
771k
  int sign_bias = 0;
1785
1786
771k
  int intra_rd_penalty =
1787
771k
      10 * vp8_dc_quant(cpi->common.base_qindex, cpi->common.y1dc_delta_q);
1788
1789
771k
#if CONFIG_TEMPORAL_DENOISING
1790
771k
  unsigned int zero_mv_sse = UINT_MAX, best_sse = UINT_MAX,
1791
771k
               best_rd_sse = UINT_MAX;
1792
771k
#endif
1793
1794
  // _uv variables are not set consistantly before calling update_best_mode.
1795
771k
  rd.rate_uv = 0;
1796
771k
  rd.distortion_uv = 0;
1797
1798
771k
  mode_mv = mode_mv_sb[sign_bias];
1799
771k
  best_ref_mv.as_int = 0;
1800
771k
  best_mode.rd = INT_MAX;
1801
771k
  best_mode.yrd = INT_MAX;
1802
771k
  best_mode.intra_rd = INT_MAX;
1803
771k
  memset(mode_mv_sb, 0, sizeof(mode_mv_sb));
1804
771k
  memset(&best_mode.mbmode, 0, sizeof(best_mode.mbmode));
1805
771k
  memset(&best_mode.bmodes, 0, sizeof(best_mode.bmodes));
1806
1807
  /* Setup search priorities */
1808
771k
  get_reference_search_order(cpi, ref_frame_map);
1809
1810
  /* Check to see if there is at least 1 valid reference frame that we need
1811
   * to calculate near_mvs.
1812
   */
1813
771k
  if (ref_frame_map[1] > 0) {
1814
771k
    sign_bias = vp8_find_near_mvs_bias(
1815
771k
        &x->e_mbd, x->e_mbd.mode_info_context, mode_mv_sb, best_ref_mv_sb,
1816
771k
        mdcounts, ref_frame_map[1], cpi->common.ref_frame_sign_bias);
1817
1818
771k
    mode_mv = mode_mv_sb[sign_bias];
1819
771k
    best_ref_mv.as_int = best_ref_mv_sb[sign_bias].as_int;
1820
771k
  }
1821
1822
771k
  get_predictor_pointers(cpi, plane, recon_yoffset, recon_uvoffset);
1823
1824
771k
  *returnintra = INT_MAX;
1825
  /* Count of the number of MBs tested so far this frame */
1826
771k
  x->mbs_tested_so_far++;
1827
1828
771k
  x->skip = 0;
1829
1830
16.2M
  for (mode_index = 0; mode_index < MAX_MODES; ++mode_index) {
1831
15.4M
    int this_rd = INT_MAX;
1832
15.4M
    int disable_skip = 0;
1833
15.4M
    int other_cost = 0;
1834
15.4M
    int this_ref_frame = ref_frame_map[vp8_ref_frame_order[mode_index]];
1835
1836
    /* Test best rd so far against threshold for trying this mode. */
1837
15.4M
    if (best_mode.rd <= x->rd_threshes[mode_index]) continue;
1838
1839
13.8M
    if (this_ref_frame < 0) continue;
1840
1841
    /* These variables hold are rolling total cost and distortion for
1842
     * this mode
1843
     */
1844
8.58M
    rd.rate2 = 0;
1845
8.58M
    rd.distortion2 = 0;
1846
1847
8.58M
    this_mode = vp8_mode_order[mode_index];
1848
1849
8.58M
    x->e_mbd.mode_info_context->mbmi.mode = this_mode;
1850
8.58M
    x->e_mbd.mode_info_context->mbmi.ref_frame = this_ref_frame;
1851
1852
    /* Only consider ZEROMV/ALTREF_FRAME for alt ref frame,
1853
     * unless ARNR filtering is enabled in which case we want
1854
     * an unfiltered alternative
1855
     */
1856
8.58M
    if (cpi->is_src_frame_alt_ref && (cpi->oxcf.arnr_max_frames == 0)) {
1857
0
      if (this_mode != ZEROMV ||
1858
0
          x->e_mbd.mode_info_context->mbmi.ref_frame != ALTREF_FRAME) {
1859
0
        continue;
1860
0
      }
1861
0
    }
1862
1863
    /* everything but intra */
1864
8.58M
    if (x->e_mbd.mode_info_context->mbmi.ref_frame) {
1865
5.39M
      assert(plane[this_ref_frame][0] != NULL &&
1866
5.39M
             plane[this_ref_frame][1] != NULL &&
1867
5.39M
             plane[this_ref_frame][2] != NULL);
1868
5.39M
      x->e_mbd.pre.y_buffer = plane[this_ref_frame][0];
1869
5.39M
      x->e_mbd.pre.u_buffer = plane[this_ref_frame][1];
1870
5.39M
      x->e_mbd.pre.v_buffer = plane[this_ref_frame][2];
1871
1872
5.39M
      if (sign_bias != cpi->common.ref_frame_sign_bias[this_ref_frame]) {
1873
0
        sign_bias = cpi->common.ref_frame_sign_bias[this_ref_frame];
1874
0
        mode_mv = mode_mv_sb[sign_bias];
1875
0
        best_ref_mv.as_int = best_ref_mv_sb[sign_bias].as_int;
1876
0
      }
1877
5.39M
    }
1878
1879
    /* Check to see if the testing frequency for this mode is at its
1880
     * max If so then prevent it from being tested and increase the
1881
     * threshold for its testing
1882
     */
1883
8.58M
    if (x->mode_test_hit_counts[mode_index] &&
1884
7.65M
        (cpi->mode_check_freq[mode_index] > 1)) {
1885
217k
      if (x->mbs_tested_so_far <= cpi->mode_check_freq[mode_index] *
1886
217k
                                      x->mode_test_hit_counts[mode_index]) {
1887
        /* Increase the threshold for coding this mode to make it
1888
         * less likely to be chosen
1889
         */
1890
121k
        x->rd_thresh_mult[mode_index] += 4;
1891
1892
121k
        if (x->rd_thresh_mult[mode_index] > MAX_THRESHMULT) {
1893
17.9k
          x->rd_thresh_mult[mode_index] = MAX_THRESHMULT;
1894
17.9k
        }
1895
1896
121k
        x->rd_threshes[mode_index] =
1897
121k
            (cpi->rd_baseline_thresh[mode_index] >> 7) *
1898
121k
            x->rd_thresh_mult[mode_index];
1899
1900
121k
        continue;
1901
121k
      }
1902
217k
    }
1903
1904
    /* We have now reached the point where we are going to test the
1905
     * current mode so increment the counter for the number of times
1906
     * it has been tested
1907
     */
1908
8.46M
    x->mode_test_hit_counts[mode_index]++;
1909
1910
    /* Experimental code. Special case for gf and arf zeromv modes.
1911
     * Increase zbin size to supress noise
1912
     */
1913
8.46M
    if (x->zbin_mode_boost_enabled) {
1914
0
      if (this_ref_frame == INTRA_FRAME) {
1915
0
        x->zbin_mode_boost = 0;
1916
0
      } else {
1917
0
        if (vp8_mode_order[mode_index] == ZEROMV) {
1918
0
          if (this_ref_frame != LAST_FRAME) {
1919
0
            x->zbin_mode_boost = GF_ZEROMV_ZBIN_BOOST;
1920
0
          } else {
1921
0
            x->zbin_mode_boost = LF_ZEROMV_ZBIN_BOOST;
1922
0
          }
1923
0
        } else if (vp8_mode_order[mode_index] == SPLITMV) {
1924
0
          x->zbin_mode_boost = 0;
1925
0
        } else {
1926
0
          x->zbin_mode_boost = MV_ZBIN_BOOST;
1927
0
        }
1928
0
      }
1929
1930
0
      vp8_update_zbin_extra(cpi, x);
1931
0
    }
1932
1933
8.46M
    if (!uv_intra_done && this_ref_frame == INTRA_FRAME) {
1934
771k
      rd_pick_intra_mbuv_mode(x, &uv_intra_rate, &uv_intra_rate_tokenonly,
1935
771k
                              &uv_intra_distortion);
1936
771k
      uv_intra_mode = x->e_mbd.mode_info_context->mbmi.uv_mode;
1937
1938
      /*
1939
       * Total of the eobs is used later to further adjust rate2. Since uv
1940
       * block's intra eobs will be overwritten when we check inter modes,
1941
       * we need to save uv_intra_tteob here.
1942
       */
1943
6.94M
      for (i = 16; i < 24; ++i) uv_intra_tteob += x->e_mbd.eobs[i];
1944
1945
771k
      uv_intra_done = 1;
1946
771k
    }
1947
1948
8.46M
    switch (this_mode) {
1949
541k
      case B_PRED: {
1950
541k
        int tmp_rd;
1951
1952
        /* Note the rate value returned here includes the cost of
1953
         * coding the BPRED mode: x->mbmode_cost[x->e_mbd.frame_type][BPRED]
1954
         */
1955
541k
        int distortion;
1956
541k
        tmp_rd = rd_pick_intra4x4mby_modes(x, &rate, &rd.rate_y, &distortion,
1957
541k
                                           best_mode.yrd);
1958
541k
        rd.rate2 += rate;
1959
541k
        rd.distortion2 += distortion;
1960
1961
541k
        if (tmp_rd < best_mode.yrd) {
1962
217k
          assert(uv_intra_done);
1963
217k
          rd.rate2 += uv_intra_rate;
1964
217k
          rd.rate_uv = uv_intra_rate_tokenonly;
1965
217k
          rd.distortion2 += uv_intra_distortion;
1966
217k
          rd.distortion_uv = uv_intra_distortion;
1967
324k
        } else {
1968
324k
          this_rd = INT_MAX;
1969
324k
          disable_skip = 1;
1970
324k
        }
1971
541k
        break;
1972
0
      }
1973
1974
730k
      case SPLITMV: {
1975
730k
        int tmp_rd;
1976
730k
        int this_rd_thresh;
1977
730k
        int distortion;
1978
1979
730k
        this_rd_thresh = (vp8_ref_frame_order[mode_index] == 1)
1980
730k
                             ? x->rd_threshes[THR_NEW1]
1981
730k
                             : x->rd_threshes[THR_NEW3];
1982
730k
        this_rd_thresh = (vp8_ref_frame_order[mode_index] == 2)
1983
730k
                             ? x->rd_threshes[THR_NEW2]
1984
730k
                             : this_rd_thresh;
1985
1986
730k
        tmp_rd = vp8_rd_pick_best_mbsegmentation(
1987
730k
            cpi, x, &best_ref_mv, best_mode.yrd, mdcounts, &rate, &rd.rate_y,
1988
730k
            &distortion, this_rd_thresh);
1989
1990
730k
        rd.rate2 += rate;
1991
730k
        rd.distortion2 += distortion;
1992
1993
        /* If even the 'Y' rd value of split is higher than best so far
1994
         * then don't bother looking at UV
1995
         */
1996
730k
        if (tmp_rd < best_mode.yrd) {
1997
          /* Now work out UV cost and add it in */
1998
386k
          rd_inter4x4_uv(cpi, x, &rd.rate_uv, &rd.distortion_uv,
1999
386k
                         cpi->common.full_pixel);
2000
386k
          rd.rate2 += rd.rate_uv;
2001
386k
          rd.distortion2 += rd.distortion_uv;
2002
386k
        } else {
2003
344k
          this_rd = INT_MAX;
2004
344k
          disable_skip = 1;
2005
344k
        }
2006
730k
        break;
2007
0
      }
2008
771k
      case DC_PRED:
2009
1.39M
      case V_PRED:
2010
2.02M
      case H_PRED:
2011
2.64M
      case TM_PRED: {
2012
2.64M
        int distortion;
2013
2.64M
        x->e_mbd.mode_info_context->mbmi.ref_frame = INTRA_FRAME;
2014
2015
2.64M
        vp8_build_intra_predictors_mby_s(
2016
2.64M
            xd, xd->dst.y_buffer - xd->dst.y_stride, xd->dst.y_buffer - 1,
2017
2.64M
            xd->dst.y_stride, xd->predictor, 16);
2018
2.64M
        macro_block_yrd(x, &rd.rate_y, &distortion);
2019
2.64M
        rd.rate2 += rd.rate_y;
2020
2.64M
        rd.distortion2 += distortion;
2021
2.64M
        rd.rate2 += x->mbmode_cost[x->e_mbd.frame_type]
2022
2.64M
                                  [x->e_mbd.mode_info_context->mbmi.mode];
2023
2.64M
        assert(uv_intra_done);
2024
2.64M
        rd.rate2 += uv_intra_rate;
2025
2.64M
        rd.rate_uv = uv_intra_rate_tokenonly;
2026
2.64M
        rd.distortion2 += uv_intra_distortion;
2027
2.64M
        rd.distortion_uv = uv_intra_distortion;
2028
2.64M
        break;
2029
2.02M
      }
2030
2031
981k
      case NEWMV: {
2032
981k
        int thissme;
2033
981k
        int bestsme = INT_MAX;
2034
981k
        int step_param = cpi->sf.first_step;
2035
981k
        int further_steps;
2036
981k
        int n;
2037
        /* If last step (1-away) of n-step search doesn't pick the center point
2038
           as the best match, we will do a final 1-away diamond refining search
2039
        */
2040
981k
        int do_refine = 1;
2041
2042
981k
        int sadpb = x->sadperbit16;
2043
981k
        int_mv mvp_full;
2044
2045
981k
        int col_min = ((best_ref_mv.as_mv.col + 7) >> 3) - MAX_FULL_PEL_VAL;
2046
981k
        int row_min = ((best_ref_mv.as_mv.row + 7) >> 3) - MAX_FULL_PEL_VAL;
2047
981k
        int col_max = (best_ref_mv.as_mv.col >> 3) + MAX_FULL_PEL_VAL;
2048
981k
        int row_max = (best_ref_mv.as_mv.row >> 3) + MAX_FULL_PEL_VAL;
2049
2050
981k
        int tmp_col_min = x->mv_col_min;
2051
981k
        int tmp_col_max = x->mv_col_max;
2052
981k
        int tmp_row_min = x->mv_row_min;
2053
981k
        int tmp_row_max = x->mv_row_max;
2054
2055
981k
        if (!saddone) {
2056
651k
          vp8_cal_sad(cpi, xd, x, recon_yoffset, &near_sadidx[0]);
2057
651k
          saddone = 1;
2058
651k
        }
2059
2060
981k
        vp8_mv_pred(cpi, &x->e_mbd, x->e_mbd.mode_info_context, &mvp,
2061
981k
                    x->e_mbd.mode_info_context->mbmi.ref_frame,
2062
981k
                    cpi->common.ref_frame_sign_bias, &sr, &near_sadidx[0]);
2063
2064
981k
        mvp_full.as_mv.col = mvp.as_mv.col >> 3;
2065
981k
        mvp_full.as_mv.row = mvp.as_mv.row >> 3;
2066
2067
        /* Get intersection of UMV window and valid MV window to
2068
         * reduce # of checks in diamond search.
2069
         */
2070
981k
        if (x->mv_col_min < col_min) x->mv_col_min = col_min;
2071
981k
        if (x->mv_col_max > col_max) x->mv_col_max = col_max;
2072
981k
        if (x->mv_row_min < row_min) x->mv_row_min = row_min;
2073
981k
        if (x->mv_row_max > row_max) x->mv_row_max = row_max;
2074
2075
        /* adjust search range according to sr from mv prediction */
2076
981k
        if (sr > step_param) step_param = sr;
2077
2078
        /* Initial step/diamond search */
2079
981k
        {
2080
981k
          bestsme = cpi->diamond_search_sad(
2081
981k
              x, b, d, &mvp_full, &d->bmi.mv, step_param, sadpb, &num00,
2082
981k
              &cpi->fn_ptr[BLOCK_16X16], x->mvcost, &best_ref_mv);
2083
981k
          mode_mv[NEWMV].as_int = d->bmi.mv.as_int;
2084
2085
          /* Further step/diamond searches as necessary */
2086
981k
          further_steps = (cpi->sf.max_step_search_steps - 1) - step_param;
2087
2088
981k
          n = num00;
2089
981k
          num00 = 0;
2090
2091
          /* If there won't be more n-step search, check to see if refining
2092
           * search is needed. */
2093
981k
          if (n > further_steps) do_refine = 0;
2094
2095
4.32M
          while (n < further_steps) {
2096
3.34M
            n++;
2097
2098
3.34M
            if (num00) {
2099
297k
              num00--;
2100
3.04M
            } else {
2101
3.04M
              thissme = cpi->diamond_search_sad(
2102
3.04M
                  x, b, d, &mvp_full, &d->bmi.mv, step_param + n, sadpb, &num00,
2103
3.04M
                  &cpi->fn_ptr[BLOCK_16X16], x->mvcost, &best_ref_mv);
2104
2105
              /* check to see if refining search is needed. */
2106
3.04M
              if (num00 > (further_steps - n)) do_refine = 0;
2107
2108
3.04M
              if (thissme < bestsme) {
2109
481k
                bestsme = thissme;
2110
481k
                mode_mv[NEWMV].as_int = d->bmi.mv.as_int;
2111
2.56M
              } else {
2112
2.56M
                d->bmi.mv.as_int = mode_mv[NEWMV].as_int;
2113
2.56M
              }
2114
3.04M
            }
2115
3.34M
          }
2116
981k
        }
2117
2118
        /* final 1-away diamond refining search */
2119
981k
        if (do_refine == 1) {
2120
654k
          int search_range;
2121
2122
654k
          search_range = 8;
2123
2124
654k
          thissme = cpi->refining_search_sad(
2125
654k
              x, b, d, &d->bmi.mv, sadpb, search_range,
2126
654k
              &cpi->fn_ptr[BLOCK_16X16], x->mvcost, &best_ref_mv);
2127
2128
654k
          if (thissme < bestsme) {
2129
28.2k
            bestsme = thissme;
2130
28.2k
            mode_mv[NEWMV].as_int = d->bmi.mv.as_int;
2131
625k
          } else {
2132
625k
            d->bmi.mv.as_int = mode_mv[NEWMV].as_int;
2133
625k
          }
2134
654k
        }
2135
2136
981k
        x->mv_col_min = tmp_col_min;
2137
981k
        x->mv_col_max = tmp_col_max;
2138
981k
        x->mv_row_min = tmp_row_min;
2139
981k
        x->mv_row_max = tmp_row_max;
2140
2141
981k
        if (bestsme < INT_MAX) {
2142
981k
          int dis; /* TODO: use dis in distortion calculation later. */
2143
981k
          unsigned int sse;
2144
981k
          cpi->find_fractional_mv_step(
2145
981k
              x, b, d, &d->bmi.mv, &best_ref_mv, x->errorperbit,
2146
981k
              &cpi->fn_ptr[BLOCK_16X16], x->mvcost, &dis, &sse);
2147
981k
        }
2148
2149
981k
        mode_mv[NEWMV].as_int = d->bmi.mv.as_int;
2150
2151
        /* Add the new motion vector cost to our rolling cost variable */
2152
981k
        rd.rate2 +=
2153
981k
            vp8_mv_bit_cost(&mode_mv[NEWMV], &best_ref_mv, x->mvcost, 96);
2154
981k
      }
2155
        // fall through
2156
2157
2.16M
      case NEARESTMV:
2158
3.35M
      case NEARMV:
2159
        /* Clip "next_nearest" so that it does not extend to far out
2160
         * of image
2161
         */
2162
3.35M
        vp8_clamp_mv2(&mode_mv[this_mode], xd);
2163
2164
        /* Do not bother proceeding if the vector (from newmv, nearest
2165
         * or near) is 0,0 as this should then be coded using the zeromv
2166
         * mode.
2167
         */
2168
3.35M
        if (((this_mode == NEARMV) || (this_mode == NEARESTMV)) &&
2169
2.37M
            (mode_mv[this_mode].as_int == 0)) {
2170
1.69M
          continue;
2171
1.69M
        }
2172
        // fall through
2173
2174
2.84M
      case ZEROMV:
2175
2176
        /* Trap vectors that reach beyond the UMV borders
2177
         * Note that ALL New MV, Nearest MV Near MV and Zero MV code
2178
         * drops through to this point because of the lack of break
2179
         * statements in the previous two cases.
2180
         */
2181
2.84M
        if (((mode_mv[this_mode].as_mv.row >> 3) < x->mv_row_min) ||
2182
2.84M
            ((mode_mv[this_mode].as_mv.row >> 3) > x->mv_row_max) ||
2183
2.84M
            ((mode_mv[this_mode].as_mv.col >> 3) < x->mv_col_min) ||
2184
2.84M
            ((mode_mv[this_mode].as_mv.col >> 3) > x->mv_col_max)) {
2185
0
          continue;
2186
0
        }
2187
2188
2.84M
        vp8_set_mbmode_and_mvs(x, this_mode, &mode_mv[this_mode]);
2189
2.84M
        this_rd = evaluate_inter_mode_rd(mdcounts, &rd, &disable_skip, cpi, x);
2190
2.84M
        break;
2191
2192
0
      default: break;
2193
8.46M
    }
2194
2195
6.76M
    this_rd =
2196
6.76M
        calculate_final_rd_costs(this_rd, &rd, &other_cost, disable_skip,
2197
6.76M
                                 uv_intra_tteob, intra_rd_penalty, cpi, x);
2198
2199
    /* Keep record of best intra distortion */
2200
6.76M
    if ((x->e_mbd.mode_info_context->mbmi.ref_frame == INTRA_FRAME) &&
2201
3.18M
        (this_rd < best_mode.intra_rd)) {
2202
1.14M
      best_mode.intra_rd = this_rd;
2203
1.14M
      *returnintra = rd.distortion2;
2204
1.14M
    }
2205
6.76M
#if CONFIG_TEMPORAL_DENOISING
2206
6.76M
    if (cpi->oxcf.noise_sensitivity) {
2207
0
      unsigned int sse;
2208
0
      vp8_get_inter_mbpred_error(x, &cpi->fn_ptr[BLOCK_16X16], &sse,
2209
0
                                 mode_mv[this_mode]);
2210
2211
0
      if (sse < best_rd_sse) best_rd_sse = sse;
2212
2213
      /* Store for later use by denoiser. */
2214
0
      if (this_mode == ZEROMV && sse < zero_mv_sse) {
2215
0
        zero_mv_sse = sse;
2216
0
        x->best_zeromv_reference_frame =
2217
0
            x->e_mbd.mode_info_context->mbmi.ref_frame;
2218
0
      }
2219
2220
      /* Store the best NEWMV in x for later use in the denoiser. */
2221
0
      if (x->e_mbd.mode_info_context->mbmi.mode == NEWMV && sse < best_sse) {
2222
0
        best_sse = sse;
2223
0
        vp8_get_inter_mbpred_error(x, &cpi->fn_ptr[BLOCK_16X16], &best_sse,
2224
0
                                   mode_mv[this_mode]);
2225
0
        x->best_sse_inter_mode = NEWMV;
2226
0
        x->best_sse_mv = x->e_mbd.mode_info_context->mbmi.mv;
2227
0
        x->need_to_clamp_best_mvs =
2228
0
            x->e_mbd.mode_info_context->mbmi.need_to_clamp_mvs;
2229
0
        x->best_reference_frame = x->e_mbd.mode_info_context->mbmi.ref_frame;
2230
0
      }
2231
0
    }
2232
6.76M
#endif
2233
2234
    /* Did this mode help.. i.i is it the new best mode */
2235
6.76M
    if (this_rd < best_mode.rd || x->skip) {
2236
      /* Note index of best mode so far */
2237
2.51M
      best_mode_index = mode_index;
2238
2.51M
      *returnrate = rd.rate2;
2239
2.51M
      *returndistortion = rd.distortion2;
2240
2.51M
      if (this_mode <= B_PRED) {
2241
1.01M
        x->e_mbd.mode_info_context->mbmi.uv_mode = uv_intra_mode;
2242
        /* required for left and above block mv */
2243
1.01M
        x->e_mbd.mode_info_context->mbmi.mv.as_int = 0;
2244
1.01M
      }
2245
2.51M
      update_best_mode(&best_mode, this_rd, &rd, other_cost, x);
2246
2247
      /* Testing this mode gave rise to an improvement in best error
2248
       * score. Lower threshold a bit for next time
2249
       */
2250
2.51M
      x->rd_thresh_mult[mode_index] =
2251
2.51M
          (x->rd_thresh_mult[mode_index] >= (MIN_THRESHMULT + 2))
2252
2.51M
              ? x->rd_thresh_mult[mode_index] - 2
2253
2.51M
              : MIN_THRESHMULT;
2254
2.51M
    }
2255
2256
    /* If the mode did not help improve the best error case then raise
2257
     * the threshold for testing that mode next time around.
2258
     */
2259
4.25M
    else {
2260
4.25M
      x->rd_thresh_mult[mode_index] += 4;
2261
2262
4.25M
      if (x->rd_thresh_mult[mode_index] > MAX_THRESHMULT) {
2263
2.10M
        x->rd_thresh_mult[mode_index] = MAX_THRESHMULT;
2264
2.10M
      }
2265
4.25M
    }
2266
6.76M
    x->rd_threshes[mode_index] = (cpi->rd_baseline_thresh[mode_index] >> 7) *
2267
6.76M
                                 x->rd_thresh_mult[mode_index];
2268
2269
6.76M
    if (x->skip) break;
2270
6.76M
  }
2271
2272
  /* Reduce the activation RD thresholds for the best choice mode */
2273
771k
  if ((cpi->rd_baseline_thresh[best_mode_index] > 0) &&
2274
508k
      (cpi->rd_baseline_thresh[best_mode_index] < (INT_MAX >> 2))) {
2275
508k
    int best_adjustment = (x->rd_thresh_mult[best_mode_index] >> 2);
2276
2277
508k
    x->rd_thresh_mult[best_mode_index] =
2278
508k
        (x->rd_thresh_mult[best_mode_index] >=
2279
508k
         (MIN_THRESHMULT + best_adjustment))
2280
508k
            ? x->rd_thresh_mult[best_mode_index] - best_adjustment
2281
508k
            : MIN_THRESHMULT;
2282
508k
    x->rd_threshes[best_mode_index] =
2283
508k
        (cpi->rd_baseline_thresh[best_mode_index] >> 7) *
2284
508k
        x->rd_thresh_mult[best_mode_index];
2285
508k
  }
2286
2287
771k
#if CONFIG_TEMPORAL_DENOISING
2288
771k
  if (cpi->oxcf.noise_sensitivity) {
2289
0
    int block_index = mb_row * cpi->common.mb_cols + mb_col;
2290
0
    if (x->best_sse_inter_mode == DC_PRED) {
2291
      /* No best MV found. */
2292
0
      x->best_sse_inter_mode = best_mode.mbmode.mode;
2293
0
      x->best_sse_mv = best_mode.mbmode.mv;
2294
0
      x->need_to_clamp_best_mvs = best_mode.mbmode.need_to_clamp_mvs;
2295
0
      x->best_reference_frame = best_mode.mbmode.ref_frame;
2296
0
      best_sse = best_rd_sse;
2297
0
    }
2298
0
    vp8_denoiser_denoise_mb(&cpi->denoiser, x, best_sse, zero_mv_sse,
2299
0
                            recon_yoffset, recon_uvoffset, &cpi->common.lf_info,
2300
0
                            mb_row, mb_col, block_index, 0);
2301
2302
    /* Reevaluate ZEROMV after denoising. */
2303
0
    if (best_mode.mbmode.ref_frame == INTRA_FRAME &&
2304
0
        x->best_zeromv_reference_frame != INTRA_FRAME) {
2305
0
      int this_rd = INT_MAX;
2306
0
      int disable_skip = 0;
2307
0
      int other_cost = 0;
2308
0
      int this_ref_frame = x->best_zeromv_reference_frame;
2309
0
      rd.rate2 =
2310
0
          x->ref_frame_cost[this_ref_frame] + vp8_cost_mv_ref(ZEROMV, mdcounts);
2311
0
      rd.distortion2 = 0;
2312
2313
      /* set up the proper prediction buffers for the frame */
2314
0
      x->e_mbd.mode_info_context->mbmi.ref_frame = this_ref_frame;
2315
0
      x->e_mbd.pre.y_buffer = plane[this_ref_frame][0];
2316
0
      x->e_mbd.pre.u_buffer = plane[this_ref_frame][1];
2317
0
      x->e_mbd.pre.v_buffer = plane[this_ref_frame][2];
2318
2319
0
      x->e_mbd.mode_info_context->mbmi.mode = ZEROMV;
2320
0
      x->e_mbd.mode_info_context->mbmi.uv_mode = DC_PRED;
2321
0
      x->e_mbd.mode_info_context->mbmi.mv.as_int = 0;
2322
2323
0
      this_rd = evaluate_inter_mode_rd(mdcounts, &rd, &disable_skip, cpi, x);
2324
0
      this_rd =
2325
0
          calculate_final_rd_costs(this_rd, &rd, &other_cost, disable_skip,
2326
0
                                   uv_intra_tteob, intra_rd_penalty, cpi, x);
2327
0
      if (this_rd < best_mode.rd || x->skip) {
2328
0
        *returnrate = rd.rate2;
2329
0
        *returndistortion = rd.distortion2;
2330
0
        update_best_mode(&best_mode, this_rd, &rd, other_cost, x);
2331
0
      }
2332
0
    }
2333
0
  }
2334
771k
#endif
2335
2336
771k
  if (cpi->is_src_frame_alt_ref &&
2337
0
      (best_mode.mbmode.mode != ZEROMV ||
2338
0
       best_mode.mbmode.ref_frame != ALTREF_FRAME)) {
2339
0
    x->e_mbd.mode_info_context->mbmi.mode = ZEROMV;
2340
0
    x->e_mbd.mode_info_context->mbmi.ref_frame = ALTREF_FRAME;
2341
0
    x->e_mbd.mode_info_context->mbmi.mv.as_int = 0;
2342
0
    x->e_mbd.mode_info_context->mbmi.uv_mode = DC_PRED;
2343
0
    x->e_mbd.mode_info_context->mbmi.mb_skip_coeff =
2344
0
        (cpi->common.mb_no_coeff_skip);
2345
0
    x->e_mbd.mode_info_context->mbmi.partitioning = 0;
2346
0
    return;
2347
0
  }
2348
2349
  /* macroblock modes */
2350
771k
  x->e_mbd.mode_info_context->mbmi = best_mode.mbmode;
2351
2352
771k
  if (best_mode.mbmode.mode == B_PRED) {
2353
3.56M
    for (i = 0; i < 16; ++i) {
2354
3.35M
      xd->mode_info_context->bmi[i].as_mode = best_mode.bmodes[i].as_mode;
2355
3.35M
    }
2356
209k
  }
2357
2358
771k
  if (best_mode.mbmode.mode == SPLITMV) {
2359
2.94M
    for (i = 0; i < 16; ++i) {
2360
2.77M
      xd->mode_info_context->bmi[i].mv.as_int = best_mode.bmodes[i].mv.as_int;
2361
2.77M
    }
2362
2363
173k
    *x->partition_info = best_mode.partition;
2364
2365
173k
    x->e_mbd.mode_info_context->mbmi.mv.as_int =
2366
173k
        x->partition_info->bmi[15].mv.as_int;
2367
173k
  }
2368
2369
771k
  if (sign_bias !=
2370
771k
      cpi->common.ref_frame_sign_bias[xd->mode_info_context->mbmi.ref_frame]) {
2371
0
    best_ref_mv.as_int = best_ref_mv_sb[!sign_bias].as_int;
2372
0
  }
2373
2374
771k
  rd_update_mvcount(x, &best_ref_mv);
2375
771k
}
2376
2377
522k
void vp8_rd_pick_intra_mode(MACROBLOCK *x, int *rate) {
2378
522k
  int error4x4, error16x16;
2379
522k
  int rate4x4, rate16x16 = 0, rateuv;
2380
522k
  int dist4x4, dist16x16, distuv;
2381
522k
  int rate_;
2382
522k
  int rate4x4_tokenonly = 0;
2383
522k
  int rate16x16_tokenonly = 0;
2384
522k
  int rateuv_tokenonly = 0;
2385
2386
522k
  x->e_mbd.mode_info_context->mbmi.ref_frame = INTRA_FRAME;
2387
2388
522k
  rd_pick_intra_mbuv_mode(x, &rateuv, &rateuv_tokenonly, &distuv);
2389
522k
  rate_ = rateuv;
2390
2391
522k
  error16x16 = rd_pick_intra16x16mby_mode(x, &rate16x16, &rate16x16_tokenonly,
2392
522k
                                          &dist16x16);
2393
2394
522k
  error4x4 = rd_pick_intra4x4mby_modes(x, &rate4x4, &rate4x4_tokenonly,
2395
522k
                                       &dist4x4, error16x16);
2396
2397
522k
  if (error4x4 < error16x16) {
2398
274k
    x->e_mbd.mode_info_context->mbmi.mode = B_PRED;
2399
274k
    rate_ += rate4x4;
2400
274k
  } else {
2401
247k
    rate_ += rate16x16;
2402
247k
  }
2403
2404
522k
  *rate = rate_;
2405
522k
}