Coverage Report

Created: 2026-09-03 06:27

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/libavif/ext/aom/av1/encoder/pickcdef.c
Line
Count
Source
1
/*
2
 * Copyright (c) 2016, Alliance for Open Media. All rights reserved.
3
 *
4
 * This source code is subject to the terms of the BSD 2 Clause License and
5
 * the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
6
 * was not distributed with this source code in the LICENSE file, you can
7
 * obtain it at www.aomedia.org/license/software. If the Alliance for Open
8
 * Media Patent License 1.0 was not distributed with this source code in the
9
 * PATENTS file, you can obtain it at www.aomedia.org/license/patent.
10
 */
11
12
#include <math.h>
13
#include <stdbool.h>
14
#include <stddef.h>
15
#include <string.h>
16
17
#include "config/aom_dsp_rtcd.h"
18
#include "config/aom_scale_rtcd.h"
19
20
#include "aom/aom_integer.h"
21
#include "av1/common/av1_common_int.h"
22
#include "av1/common/reconinter.h"
23
#include "av1/encoder/encoder.h"
24
#include "av1/encoder/ethread.h"
25
#include "av1/encoder/pickcdef.h"
26
27
// Get primary and secondary filter strength for the given strength index and
28
// search method
29
static inline void get_cdef_filter_strengths(CDEF_PICK_METHOD pick_method,
30
                                             int *pri_strength,
31
                                             int *sec_strength,
32
1.68M
                                             int strength_idx) {
33
1.68M
  const int tot_sec_filter =
34
1.68M
      (pick_method == CDEF_FAST_SEARCH_LVL5)
35
1.68M
          ? REDUCED_SEC_STRENGTHS_LVL5
36
1.68M
          : ((pick_method >= CDEF_FAST_SEARCH_LVL3) ? REDUCED_SEC_STRENGTHS_LVL3
37
18.4E
                                                    : CDEF_SEC_STRENGTHS);
38
1.68M
  const int pri_idx = strength_idx / tot_sec_filter;
39
1.68M
  const int sec_idx = strength_idx % tot_sec_filter;
40
1.68M
  *pri_strength = pri_idx;
41
1.68M
  *sec_strength = sec_idx;
42
1.68M
  if (pick_method == CDEF_FULL_SEARCH) return;
43
44
1.68M
  switch (pick_method) {
45
0
    case CDEF_FAST_SEARCH_LVL1:
46
0
      assert(pri_idx < REDUCED_PRI_STRENGTHS_LVL1);
47
0
      *pri_strength = priconv_lvl1[pri_idx];
48
0
      break;
49
0
    case CDEF_FAST_SEARCH_LVL2:
50
0
      assert(pri_idx < REDUCED_PRI_STRENGTHS_LVL2);
51
0
      *pri_strength = priconv_lvl2[pri_idx];
52
0
      break;
53
0
    case CDEF_FAST_SEARCH_LVL3:
54
0
      assert(pri_idx < REDUCED_PRI_STRENGTHS_LVL2);
55
0
      assert(sec_idx < REDUCED_SEC_STRENGTHS_LVL3);
56
0
      *pri_strength = priconv_lvl2[pri_idx];
57
0
      *sec_strength = secconv_lvl3[sec_idx];
58
0
      break;
59
1.68M
    case CDEF_FAST_SEARCH_LVL4:
60
1.68M
      assert(pri_idx < REDUCED_PRI_STRENGTHS_LVL4);
61
1.68M
      assert(sec_idx < REDUCED_SEC_STRENGTHS_LVL3);
62
1.68M
      *pri_strength = priconv_lvl4[pri_idx];
63
1.68M
      *sec_strength = secconv_lvl3[sec_idx];
64
1.68M
      break;
65
0
    case CDEF_FAST_SEARCH_LVL5:
66
0
      assert(pri_idx < REDUCED_PRI_STRENGTHS_LVL4);
67
0
      assert(sec_idx < REDUCED_SEC_STRENGTHS_LVL5);
68
0
      *pri_strength = priconv_lvl5[pri_idx];
69
0
      *sec_strength = secconv_lvl5[sec_idx];
70
0
      break;
71
0
    default: assert(0 && "Invalid CDEF search method");
72
1.68M
  }
73
1.68M
}
74
75
// Store CDEF filter strength calculated from strength index for given search
76
// method
77
#define STORE_CDEF_FILTER_STRENGTH(cdef_strength, pick_method, strength_idx) \
78
118k
  do {                                                                       \
79
118k
    get_cdef_filter_strengths((pick_method), &pri_strength, &sec_strength,   \
80
118k
                              (strength_idx));                               \
81
118k
    cdef_strength = pri_strength * CDEF_SEC_STRENGTHS + sec_strength;        \
82
118k
  } while (0)
83
84
/* Search for the best strength to add as an option, knowing we
85
   already selected nb_strengths options. */
86
static uint64_t search_one(int *lev, int nb_strengths,
87
                           uint64_t mse[][TOTAL_STRENGTHS], int sb_count,
88
237k
                           CDEF_PICK_METHOD pick_method) {
89
237k
  uint64_t tot_mse[TOTAL_STRENGTHS];
90
237k
  const int total_strengths = nb_cdef_strengths[pick_method];
91
237k
  int i, j;
92
237k
  uint64_t best_tot_mse = (uint64_t)1 << 63;
93
237k
  int best_id = 0;
94
237k
  memset(tot_mse, 0, sizeof(tot_mse));
95
965k
  for (i = 0; i < sb_count; i++) {
96
727k
    int gi;
97
727k
    uint64_t best_mse = (uint64_t)1 << 63;
98
    /* Find best mse among already selected options. */
99
1.47M
    for (gi = 0; gi < nb_strengths; gi++) {
100
748k
      if (mse[i][lev[gi]] < best_mse) {
101
486k
        best_mse = mse[i][lev[gi]];
102
486k
      }
103
748k
    }
104
    /* Find best mse when adding each possible new option. */
105
3.63M
    for (j = 0; j < total_strengths; j++) {
106
2.91M
      uint64_t best = best_mse;
107
2.91M
      if (mse[i][j] < best) best = mse[i][j];
108
2.91M
      tot_mse[j] += best;
109
2.91M
    }
110
727k
  }
111
1.18M
  for (j = 0; j < total_strengths; j++) {
112
951k
    if (tot_mse[j] < best_tot_mse) {
113
400k
      best_tot_mse = tot_mse[j];
114
400k
      best_id = j;
115
400k
    }
116
951k
  }
117
237k
  lev[nb_strengths] = best_id;
118
237k
  return best_tot_mse;
119
237k
}
120
121
/* Search for the best luma+chroma strength to add as an option, knowing we
122
   already selected nb_strengths options. */
123
static uint64_t search_one_dual(int *lev0, int *lev1, int nb_strengths,
124
                                uint64_t (**mse)[TOTAL_STRENGTHS], int sb_count,
125
2.00M
                                CDEF_PICK_METHOD pick_method) {
126
2.00M
  uint64_t tot_mse[TOTAL_STRENGTHS][TOTAL_STRENGTHS];
127
2.00M
  int i, j;
128
2.00M
  uint64_t best_tot_mse = (uint64_t)1 << 63;
129
2.00M
  int best_id0 = 0;
130
2.00M
  int best_id1 = 0;
131
2.00M
  const int total_strengths = nb_cdef_strengths[pick_method];
132
2.00M
  memset(tot_mse, 0, sizeof(tot_mse));
133
9.07M
  for (i = 0; i < sb_count; i++) {
134
7.06M
    int gi;
135
7.06M
    uint64_t best_mse = (uint64_t)1 << 63;
136
    /* Find best mse among already selected options. */
137
37.1M
    for (gi = 0; gi < nb_strengths; gi++) {
138
30.1M
      uint64_t curr = mse[0][i][lev0[gi]];
139
30.1M
      curr += mse[1][i][lev1[gi]];
140
30.1M
      if (curr < best_mse) {
141
10.9M
        best_mse = curr;
142
10.9M
      }
143
30.1M
    }
144
    /* Find best mse when adding each possible new option. */
145
35.3M
    for (j = 0; j < total_strengths; j++) {
146
28.2M
      int k;
147
141M
      for (k = 0; k < total_strengths; k++) {
148
113M
        uint64_t best = best_mse;
149
113M
        uint64_t curr = mse[0][i][j];
150
113M
        curr += mse[1][i][k];
151
113M
        if (curr < best) best = curr;
152
113M
        tot_mse[j][k] += best;
153
113M
      }
154
28.2M
    }
155
7.06M
  }
156
10.0M
  for (j = 0; j < total_strengths; j++) {
157
8.02M
    int k;
158
40.1M
    for (k = 0; k < total_strengths; k++) {
159
32.0M
      if (tot_mse[j][k] < best_tot_mse) {
160
4.29M
        best_tot_mse = tot_mse[j][k];
161
4.29M
        best_id0 = j;
162
4.29M
        best_id1 = k;
163
4.29M
      }
164
32.0M
    }
165
8.02M
  }
166
2.00M
  lev0[nb_strengths] = best_id0;
167
2.00M
  lev1[nb_strengths] = best_id1;
168
2.00M
  return best_tot_mse;
169
2.00M
}
170
171
/* Search for the set of strengths that minimizes mse. */
172
static uint64_t joint_strength_search(int *best_lev, int nb_strengths,
173
                                      uint64_t mse[][TOTAL_STRENGTHS],
174
                                      int sb_count,
175
99.4k
                                      CDEF_PICK_METHOD pick_method) {
176
99.4k
  uint64_t best_tot_mse;
177
99.4k
  int fast = (pick_method >= CDEF_FAST_SEARCH_LVL1 &&
178
99.4k
              pick_method <= CDEF_FAST_SEARCH_LVL5);
179
99.4k
  int i;
180
99.4k
  best_tot_mse = (uint64_t)1 << 63;
181
  /* Greedy search: add one strength options at a time. */
182
337k
  for (i = 0; i < nb_strengths; i++) {
183
237k
    best_tot_mse = search_one(best_lev, i, mse, sb_count, pick_method);
184
237k
  }
185
  /* Trying to refine the greedy search by reconsidering each
186
     already-selected option. */
187
99.4k
  if (!fast) {
188
0
    for (i = 0; i < 4 * nb_strengths; i++) {
189
0
      int j;
190
0
      for (j = 0; j < nb_strengths - 1; j++) best_lev[j] = best_lev[j + 1];
191
0
      best_tot_mse =
192
0
          search_one(best_lev, nb_strengths - 1, mse, sb_count, pick_method);
193
0
    }
194
0
  }
195
99.4k
  return best_tot_mse;
196
99.4k
}
197
198
/* Search for the set of luma+chroma strengths that minimizes mse. */
199
static uint64_t joint_strength_search_dual(int *best_lev0, int *best_lev1,
200
                                           int nb_strengths,
201
                                           uint64_t (**mse)[TOTAL_STRENGTHS],
202
                                           int sb_count,
203
103k
                                           CDEF_PICK_METHOD pick_method) {
204
103k
  uint64_t best_tot_mse;
205
103k
  int i;
206
103k
  best_tot_mse = (uint64_t)1 << 63;
207
  /* Greedy search: add one strength options at a time. */
208
505k
  for (i = 0; i < nb_strengths; i++) {
209
401k
    best_tot_mse =
210
401k
        search_one_dual(best_lev0, best_lev1, i, mse, sb_count, pick_method);
211
401k
  }
212
  /* Trying to refine the greedy search by reconsidering each
213
     already-selected option. */
214
1.70M
  for (i = 0; i < 4 * nb_strengths; i++) {
215
1.60M
    int j;
216
9.17M
    for (j = 0; j < nb_strengths - 1; j++) {
217
7.56M
      best_lev0[j] = best_lev0[j + 1];
218
7.56M
      best_lev1[j] = best_lev1[j + 1];
219
7.56M
    }
220
1.60M
    best_tot_mse = search_one_dual(best_lev0, best_lev1, nb_strengths - 1, mse,
221
1.60M
                                   sb_count, pick_method);
222
1.60M
  }
223
103k
  return best_tot_mse;
224
103k
}
225
226
static inline void init_src_params(int *src_stride, int *width, int *height,
227
                                   int *width_log2, int *height_log2,
228
366k
                                   BLOCK_SIZE bsize) {
229
366k
  *src_stride = block_size_wide[bsize];
230
366k
  *width = block_size_wide[bsize];
231
366k
  *height = block_size_high[bsize];
232
366k
  *width_log2 = MI_SIZE_LOG2 + mi_size_wide_log2[bsize];
233
366k
  *height_log2 = MI_SIZE_LOG2 + mi_size_high_log2[bsize];
234
366k
}
235
#if CONFIG_AV1_HIGHBITDEPTH
236
/* Compute MSE only on the blocks we filtered. */
237
static uint64_t compute_cdef_dist_highbd(void *dst, int dstride, uint16_t *src,
238
                                         cdef_list *dlist, int cdef_count,
239
                                         BLOCK_SIZE bsize, int coeff_shift,
240
366k
                                         int row, int col) {
241
366k
  assert(bsize == BLOCK_4X4 || bsize == BLOCK_4X8 || bsize == BLOCK_8X4 ||
242
366k
         bsize == BLOCK_8X8);
243
366k
  uint64_t sum = 0;
244
366k
  int bi, bx, by;
245
366k
  uint16_t *dst16 = CONVERT_TO_SHORTPTR((uint8_t *)dst);
246
366k
  uint16_t *dst_buff = &dst16[row * dstride + col];
247
366k
  int src_stride, width, height, width_log2, height_log2;
248
366k
  init_src_params(&src_stride, &width, &height, &width_log2, &height_log2,
249
366k
                  bsize);
250
7.07M
  for (bi = 0; bi < cdef_count; bi++) {
251
6.70M
    by = dlist[bi].by;
252
6.70M
    bx = dlist[bi].bx;
253
6.70M
    sum += aom_mse_wxh_16bit_highbd(
254
6.70M
        &dst_buff[(by << height_log2) * dstride + (bx << width_log2)], dstride,
255
6.70M
        &src[bi << (height_log2 + width_log2)], src_stride, width, height);
256
6.70M
  }
257
366k
  return sum >> 2 * coeff_shift;
258
366k
}
259
#endif
260
261
// Checks dual and quad block processing is applicable for block widths 8 and 4
262
// respectively.
263
static inline int is_dual_or_quad_applicable(cdef_list *dlist, int width,
264
0
                                             int cdef_count, int bi, int iter) {
265
0
  assert(width == 8 || width == 4);
266
0
  const int blk_offset = (width == 8) ? 1 : 3;
267
0
  if ((iter + blk_offset) >= cdef_count) return 0;
268
269
0
  if (dlist[bi].by == dlist[bi + blk_offset].by &&
270
0
      dlist[bi].bx + blk_offset == dlist[bi + blk_offset].bx)
271
0
    return 1;
272
273
0
  return 0;
274
0
}
275
276
static uint64_t compute_cdef_dist(void *dst, int dstride, uint16_t *src,
277
                                  cdef_list *dlist, int cdef_count,
278
                                  BLOCK_SIZE bsize, int coeff_shift, int row,
279
0
                                  int col) {
280
0
  assert(bsize == BLOCK_4X4 || bsize == BLOCK_4X8 || bsize == BLOCK_8X4 ||
281
0
         bsize == BLOCK_8X8);
282
0
  uint64_t sum = 0;
283
0
  int bi, bx, by;
284
0
  int iter = 0;
285
0
  int inc = 1;
286
0
  uint8_t *dst8 = (uint8_t *)dst;
287
0
  uint8_t *dst_buff = &dst8[row * dstride + col];
288
0
  int src_stride, width, height, width_log2, height_log2;
289
0
  init_src_params(&src_stride, &width, &height, &width_log2, &height_log2,
290
0
                  bsize);
291
292
0
  const int num_blks = 16 / width;
293
0
  for (bi = 0; bi < cdef_count; bi += inc) {
294
0
    by = dlist[bi].by;
295
0
    bx = dlist[bi].bx;
296
0
    uint16_t *src_tmp = &src[bi << (height_log2 + width_log2)];
297
0
    uint8_t *dst_tmp =
298
0
        &dst_buff[(by << height_log2) * dstride + (bx << width_log2)];
299
300
0
    if (is_dual_or_quad_applicable(dlist, width, cdef_count, bi, iter)) {
301
0
      sum += aom_mse_16xh_16bit(dst_tmp, dstride, src_tmp, width, height);
302
0
      iter += num_blks;
303
0
      inc = num_blks;
304
0
    } else {
305
0
      sum += aom_mse_wxh_16bit(dst_tmp, dstride, src_tmp, src_stride, width,
306
0
                               height);
307
0
      iter += 1;
308
0
      inc = 1;
309
0
    }
310
0
  }
311
312
0
  return sum >> 2 * coeff_shift;
313
0
}
314
315
// Fill the boundary regions of the block with CDEF_VERY_LARGE, only if the
316
// region is outside frame boundary
317
static inline void fill_borders_for_fbs_on_frame_boundary(
318
    uint16_t *inbuf, int hfilt_size, int vfilt_size,
319
    bool is_fb_on_frm_left_boundary, bool is_fb_on_frm_right_boundary,
320
393k
    bool is_fb_on_frm_top_boundary, bool is_fb_on_frm_bottom_boundary) {
321
393k
  if (!is_fb_on_frm_left_boundary && !is_fb_on_frm_right_boundary &&
322
122k
      !is_fb_on_frm_top_boundary && !is_fb_on_frm_bottom_boundary)
323
42.9k
    return;
324
350k
  if (is_fb_on_frm_bottom_boundary) {
325
    // Fill bottom region of the block
326
207k
    const int buf_offset =
327
207k
        (vfilt_size + CDEF_VBORDER) * CDEF_BSTRIDE + CDEF_HBORDER;
328
207k
    fill_rect(&inbuf[buf_offset], CDEF_BSTRIDE, CDEF_VBORDER, hfilt_size,
329
207k
              CDEF_VERY_LARGE);
330
207k
  }
331
350k
  if (is_fb_on_frm_bottom_boundary || is_fb_on_frm_left_boundary) {
332
296k
    const int buf_offset = (vfilt_size + CDEF_VBORDER) * CDEF_BSTRIDE;
333
    // Fill bottom-left region of the block
334
296k
    fill_rect(&inbuf[buf_offset], CDEF_BSTRIDE, CDEF_VBORDER, CDEF_HBORDER,
335
296k
              CDEF_VERY_LARGE);
336
296k
  }
337
350k
  if (is_fb_on_frm_bottom_boundary || is_fb_on_frm_right_boundary) {
338
296k
    const int buf_offset =
339
296k
        (vfilt_size + CDEF_VBORDER) * CDEF_BSTRIDE + hfilt_size + CDEF_HBORDER;
340
    // Fill bottom-right region of the block
341
296k
    fill_rect(&inbuf[buf_offset], CDEF_BSTRIDE, CDEF_VBORDER, CDEF_HBORDER,
342
296k
              CDEF_VERY_LARGE);
343
296k
  }
344
350k
  if (is_fb_on_frm_top_boundary) {
345
    // Fill top region of the block
346
207k
    fill_rect(&inbuf[CDEF_HBORDER], CDEF_BSTRIDE, CDEF_VBORDER, hfilt_size,
347
207k
              CDEF_VERY_LARGE);
348
207k
  }
349
350k
  if (is_fb_on_frm_top_boundary || is_fb_on_frm_left_boundary) {
350
    // Fill top-left region of the block
351
296k
    fill_rect(inbuf, CDEF_BSTRIDE, CDEF_VBORDER, CDEF_HBORDER, CDEF_VERY_LARGE);
352
296k
  }
353
350k
  if (is_fb_on_frm_top_boundary || is_fb_on_frm_right_boundary) {
354
296k
    const int buf_offset = hfilt_size + CDEF_HBORDER;
355
    // Fill top-right region of the block
356
296k
    fill_rect(&inbuf[buf_offset], CDEF_BSTRIDE, CDEF_VBORDER, CDEF_HBORDER,
357
296k
              CDEF_VERY_LARGE);
358
296k
  }
359
350k
  if (is_fb_on_frm_left_boundary) {
360
199k
    const int buf_offset = CDEF_VBORDER * CDEF_BSTRIDE;
361
    // Fill left region of the block
362
199k
    fill_rect(&inbuf[buf_offset], CDEF_BSTRIDE, vfilt_size, CDEF_HBORDER,
363
199k
              CDEF_VERY_LARGE);
364
199k
  }
365
350k
  if (is_fb_on_frm_right_boundary) {
366
199k
    const int buf_offset = CDEF_VBORDER * CDEF_BSTRIDE;
367
    // Fill right region of the block
368
199k
    fill_rect(&inbuf[buf_offset + hfilt_size + CDEF_HBORDER], CDEF_BSTRIDE,
369
199k
              vfilt_size, CDEF_HBORDER, CDEF_VERY_LARGE);
370
199k
  }
371
350k
}
372
373
// Calculate the number of 8x8/4x4 filter units for which SSE can be calculated
374
// after CDEF filtering in single function call
375
static AOM_FORCE_INLINE int get_error_calc_width_in_filt_units(
376
    cdef_list *dlist, int cdef_count, int bi, int subsampling_x,
377
5.79M
    int subsampling_y) {
378
  // TODO(Ranjit): Extend the optimization for 422
379
5.79M
  if (subsampling_x != subsampling_y) return 1;
380
381
  // Combining more blocks seems to increase encode time due to increase in
382
  // control code
383
4.17M
  if (bi + 3 < cdef_count && dlist[bi].by == dlist[bi + 3].by &&
384
1.58M
      dlist[bi].bx + 3 == dlist[bi + 3].bx) {
385
    /* Calculate error for four 8x8/4x4 blocks using 32x8/16x4 block specific
386
     * logic if y co-ordinates match and x co-ordinates are
387
     * separated by 3 for first and fourth 8x8/4x4 blocks in dlist[]. */
388
1.58M
    return 4;
389
1.58M
  }
390
2.59M
  if (bi + 1 < cdef_count && dlist[bi].by == dlist[bi + 1].by &&
391
815k
      dlist[bi].bx + 1 == dlist[bi + 1].bx) {
392
    /* Calculate error for two 8x8/4x4 blocks using 16x8/8x4 block specific
393
     * logic if their y co-ordinates match and x co-ordinates are
394
     * separated by 1 for first and second 8x8/4x4 blocks in dlist[]. */
395
811k
    return 2;
396
811k
  }
397
1.77M
  return 1;
398
2.59M
}
399
400
// Returns the block error after CDEF filtering for a given strength
401
static inline uint64_t get_filt_error(
402
    const CdefSearchCtx *cdef_search_ctx, const struct macroblockd_plane *pd,
403
    cdef_list *dlist, int dir[CDEF_NBLOCKS][CDEF_NBLOCKS], int *dirinit,
404
    int var[CDEF_NBLOCKS][CDEF_NBLOCKS], uint16_t *in, uint8_t *ref_buffer,
405
    int ref_stride, int row, int col, int pri_strength, int sec_strength,
406
1.56M
    int cdef_count, int pli, int coeff_shift, BLOCK_SIZE bs) {
407
1.56M
  uint64_t curr_sse = 0;
408
1.56M
  const BLOCK_SIZE plane_bsize =
409
1.56M
      get_plane_block_size(bs, pd->subsampling_x, pd->subsampling_y);
410
1.56M
  const int bw_log2 = 3 - pd->subsampling_x;
411
1.56M
  const int bh_log2 = 3 - pd->subsampling_y;
412
413
  // TODO(Ranjit): Extend this optimization for HBD
414
1.56M
  if (!cdef_search_ctx->use_highbitdepth) {
415
    // If all 8x8/4x4 blocks in CDEF block need to be filtered, calculate the
416
    // error at CDEF block level
417
1.20M
    const int tot_blk_count =
418
1.20M
        (block_size_wide[plane_bsize] * block_size_high[plane_bsize]) >>
419
1.20M
        (bw_log2 + bh_log2);
420
1.20M
    if (cdef_count == tot_blk_count) {
421
422k
      const ptrdiff_t buf_offset = (ptrdiff_t)row * ref_stride + col;
422
422k
      const ptrdiff_t dst_offset = (ptrdiff_t)row * pd->dst.stride + col;
423
422k
      if (pri_strength == 0 && sec_strength == 0) {
424
        // When CDEF strength is zero, filtering is not applied. Hence
425
        // error is calculated between source and unfiltered pixels
426
105k
        curr_sse =
427
105k
            aom_sse(&ref_buffer[buf_offset], ref_stride,
428
105k
                    &pd->dst.buf[dst_offset], pd->dst.stride,
429
105k
                    block_size_wide[plane_bsize], block_size_high[plane_bsize]);
430
316k
      } else {
431
316k
        DECLARE_ALIGNED(32, uint8_t, tmp_dst8[1 << (MAX_SB_SIZE_LOG2 * 2)]);
432
433
316k
        av1_cdef_filter_fb(tmp_dst8, NULL, (1 << MAX_SB_SIZE_LOG2), in,
434
316k
                           cdef_search_ctx->xdec[pli],
435
316k
                           cdef_search_ctx->ydec[pli], dir, dirinit, var, pli,
436
316k
                           dlist, cdef_count, pri_strength,
437
316k
                           sec_strength + (sec_strength == 3),
438
316k
                           cdef_search_ctx->damping, coeff_shift);
439
316k
        curr_sse =
440
316k
            aom_sse(&ref_buffer[buf_offset], ref_stride, tmp_dst8,
441
316k
                    (1 << MAX_SB_SIZE_LOG2), block_size_wide[plane_bsize],
442
316k
                    block_size_high[plane_bsize]);
443
316k
      }
444
779k
    } else {
445
      // If few 8x8/4x4 blocks in CDEF block need to be filtered, filtering
446
      // functions produce 8-bit output and the error is calculated in 8-bit
447
      // domain
448
779k
      if (pri_strength == 0 && sec_strength == 0) {
449
195k
        int num_error_calc_filt_units = 1;
450
1.64M
        for (int bi = 0; bi < cdef_count; bi = bi + num_error_calc_filt_units) {
451
1.45M
          const uint8_t by = dlist[bi].by;
452
1.45M
          const uint8_t bx = dlist[bi].bx;
453
1.45M
          const int by_pos = by << bh_log2;
454
1.45M
          const int bx_pos = bx << bw_log2;
455
1.45M
          const ptrdiff_t buf_offset =
456
1.45M
              (ptrdiff_t)(row + by_pos) * ref_stride + (col + bx_pos);
457
1.45M
          const ptrdiff_t dst_offset =
458
1.45M
              (ptrdiff_t)(row + by_pos) * pd->dst.stride + (col + bx_pos);
459
1.45M
          num_error_calc_filt_units = get_error_calc_width_in_filt_units(
460
1.45M
              dlist, cdef_count, bi, pd->subsampling_x, pd->subsampling_y);
461
1.45M
          curr_sse +=
462
1.45M
              aom_sse(&ref_buffer[buf_offset], ref_stride,
463
1.45M
                      &pd->dst.buf[dst_offset], pd->dst.stride,
464
1.45M
                      num_error_calc_filt_units * (1 << bw_log2), 1 << bh_log2);
465
1.45M
        }
466
583k
      } else {
467
583k
        DECLARE_ALIGNED(32, uint8_t, tmp_dst8[1 << (MAX_SB_SIZE_LOG2 * 2)]);
468
583k
        av1_cdef_filter_fb(tmp_dst8, NULL, (1 << MAX_SB_SIZE_LOG2), in,
469
583k
                           cdef_search_ctx->xdec[pli],
470
583k
                           cdef_search_ctx->ydec[pli], dir, dirinit, var, pli,
471
583k
                           dlist, cdef_count, pri_strength,
472
583k
                           sec_strength + (sec_strength == 3),
473
583k
                           cdef_search_ctx->damping, coeff_shift);
474
583k
        int num_error_calc_filt_units = 1;
475
4.93M
        for (int bi = 0; bi < cdef_count; bi = bi + num_error_calc_filt_units) {
476
4.35M
          const uint8_t by = dlist[bi].by;
477
4.35M
          const uint8_t bx = dlist[bi].bx;
478
4.35M
          const int by_pos = by << bh_log2;
479
4.35M
          const int bx_pos = bx << bw_log2;
480
4.35M
          const ptrdiff_t buf_offset =
481
4.35M
              (ptrdiff_t)(row + by_pos) * ref_stride + (col + bx_pos);
482
4.35M
          const ptrdiff_t tmp_buf_offset =
483
4.35M
              by_pos * (1 << MAX_SB_SIZE_LOG2) + bx_pos;
484
4.35M
          num_error_calc_filt_units = get_error_calc_width_in_filt_units(
485
4.35M
              dlist, cdef_count, bi, pd->subsampling_x, pd->subsampling_y);
486
4.35M
          curr_sse += aom_sse(
487
4.35M
              &ref_buffer[buf_offset], ref_stride, &tmp_dst8[tmp_buf_offset],
488
4.35M
              (1 << MAX_SB_SIZE_LOG2),
489
4.35M
              num_error_calc_filt_units * (1 << bw_log2), (1 << bh_log2));
490
4.35M
        }
491
583k
      }
492
779k
    }
493
1.20M
  } else {
494
367k
    DECLARE_ALIGNED(32, uint16_t, tmp_dst[1 << (MAX_SB_SIZE_LOG2 * 2)]);
495
496
367k
    av1_cdef_filter_fb(NULL, tmp_dst, CDEF_BSTRIDE, in,
497
367k
                       cdef_search_ctx->xdec[pli], cdef_search_ctx->ydec[pli],
498
367k
                       dir, dirinit, var, pli, dlist, cdef_count, pri_strength,
499
367k
                       sec_strength + (sec_strength == 3),
500
367k
                       cdef_search_ctx->damping, coeff_shift);
501
367k
    curr_sse = cdef_search_ctx->compute_cdef_dist_fn(
502
367k
        ref_buffer, ref_stride, tmp_dst, dlist, cdef_count,
503
367k
        cdef_search_ctx->bsize[pli], coeff_shift, row, col);
504
367k
  }
505
1.56M
  return curr_sse;
506
1.56M
}
507
508
// Calculates MSE at block level.
509
// Inputs:
510
//   cdef_search_ctx: Pointer to the structure containing parameters related to
511
//   CDEF search context.
512
//   fbr: Row index in units of 64x64 block
513
//   fbc: Column index in units of 64x64 block
514
//   adaptive_cdef_mode: Speed feature to control CDEF adaptively.
515
// Returns:
516
//   Nothing will be returned. Contents of cdef_search_ctx will be modified.
517
void av1_cdef_mse_calc_block(CdefSearchCtx *cdef_search_ctx,
518
                             struct aom_internal_error_info *error_info,
519
                             int fbr, int fbc, int sb_count,
520
202k
                             int adaptive_cdef_mode) {
521
  // TODO(aomedia:3276): Pass error_info to the low-level functions as required
522
  // in future to handle error propagation.
523
202k
  (void)error_info;
524
202k
  const CommonModeInfoParams *const mi_params = cdef_search_ctx->mi_params;
525
202k
  const YV12_BUFFER_CONFIG *ref = cdef_search_ctx->ref;
526
202k
  const int coeff_shift = cdef_search_ctx->coeff_shift;
527
202k
  const int *mi_wide_l2 = cdef_search_ctx->mi_wide_l2;
528
202k
  const int *mi_high_l2 = cdef_search_ctx->mi_high_l2;
529
530
  // Declare and initialize the temporary buffers.
531
202k
  DECLARE_ALIGNED(32, uint16_t, inbuf[CDEF_INBUF_SIZE]);
532
202k
  cdef_list dlist[MI_SIZE_128X128 * MI_SIZE_128X128];
533
202k
  int dir[CDEF_NBLOCKS][CDEF_NBLOCKS] = { { 0 } };
534
202k
  int var[CDEF_NBLOCKS][CDEF_NBLOCKS] = { { 0 } };
535
202k
  uint16_t *const in = inbuf + CDEF_VBORDER * CDEF_BSTRIDE + CDEF_HBORDER;
536
202k
  int nhb = AOMMIN(MI_SIZE_64X64, mi_params->mi_cols - MI_SIZE_64X64 * fbc);
537
202k
  int nvb = AOMMIN(MI_SIZE_64X64, mi_params->mi_rows - MI_SIZE_64X64 * fbr);
538
202k
  int hb_step = 1, vb_step = 1;
539
202k
  BLOCK_SIZE bs;
540
541
202k
  const MB_MODE_INFO *const mbmi =
542
202k
      mi_params->mi_grid_base[MI_SIZE_64X64 * fbr * mi_params->mi_stride +
543
202k
                              MI_SIZE_64X64 * fbc];
544
545
202k
  uint8_t *ref_buffer[MAX_MB_PLANE] = { ref->y_buffer, ref->u_buffer,
546
202k
                                        ref->v_buffer };
547
202k
  int ref_stride[MAX_MB_PLANE] = { ref->y_stride, ref->uv_stride,
548
202k
                                   ref->uv_stride };
549
550
202k
  if (mbmi->bsize == BLOCK_128X128 || mbmi->bsize == BLOCK_128X64 ||
551
202k
      mbmi->bsize == BLOCK_64X128) {
552
0
    bs = mbmi->bsize;
553
0
    if (bs == BLOCK_128X128 || bs == BLOCK_128X64) {
554
0
      nhb = AOMMIN(MI_SIZE_128X128, mi_params->mi_cols - MI_SIZE_64X64 * fbc);
555
0
      hb_step = 2;
556
0
    }
557
0
    if (bs == BLOCK_128X128 || bs == BLOCK_64X128) {
558
0
      nvb = AOMMIN(MI_SIZE_128X128, mi_params->mi_rows - MI_SIZE_64X64 * fbr);
559
0
      vb_step = 2;
560
0
    }
561
202k
  } else {
562
202k
    bs = BLOCK_64X64;
563
202k
  }
564
  // Get number of 8x8 blocks which are not skip. Cdef processing happens for
565
  // 8x8 blocks which are not skip.
566
202k
  const int cdef_count = av1_cdef_compute_sb_list(
567
202k
      mi_params, fbr * MI_SIZE_64X64, fbc * MI_SIZE_64X64, dlist, bs);
568
202k
  const bool is_fb_on_frm_left_boundary = (fbc == 0);
569
202k
  const bool is_fb_on_frm_right_boundary =
570
202k
      (fbc + hb_step == cdef_search_ctx->nhfb);
571
202k
  const bool is_fb_on_frm_top_boundary = (fbr == 0);
572
202k
  const bool is_fb_on_frm_bottom_boundary =
573
202k
      (fbr + vb_step == cdef_search_ctx->nvfb);
574
202k
  const int yoff = CDEF_VBORDER * (!is_fb_on_frm_top_boundary);
575
202k
  const int xoff = CDEF_HBORDER * (!is_fb_on_frm_left_boundary);
576
202k
  int dirinit = 0;
577
595k
  for (int pli = 0; pli < cdef_search_ctx->num_planes; pli++) {
578
    // To disable CDEF filter of chroma, set MSE of chroma strength index 0 to
579
    // zero and all non-zero strength indices to 1, such that the joint
580
    // luma-chroma strength search always chooses filter strength 0 for chroma
581
    // due to least cost.
582
393k
    if (adaptive_cdef_mode > 0 && pli > 0) {
583
0
      cdef_search_ctx->mse[pli][sb_count][0] = 0;
584
0
      for (int gi = 1; gi < cdef_search_ctx->total_strengths; gi++)
585
0
        cdef_search_ctx->mse[pli][sb_count][gi] = 1;
586
0
      break;
587
0
    }
588
589
    /* We avoid filtering the pixels for which some of the pixels to
590
    average are outside the frame. We could change the filter instead,
591
    but it would add special cases for any future vectorization. */
592
393k
    const int hfilt_size = (nhb << mi_wide_l2[pli]);
593
393k
    const int vfilt_size = (nvb << mi_high_l2[pli]);
594
393k
    const int ysize =
595
393k
        vfilt_size + CDEF_VBORDER * (!is_fb_on_frm_bottom_boundary) + yoff;
596
393k
    const int xsize =
597
393k
        hfilt_size + CDEF_HBORDER * (!is_fb_on_frm_right_boundary) + xoff;
598
393k
    const int row = fbr * MI_SIZE_64X64 << mi_high_l2[pli];
599
393k
    const int col = fbc * MI_SIZE_64X64 << mi_wide_l2[pli];
600
393k
    struct macroblockd_plane pd = cdef_search_ctx->plane[pli];
601
393k
    cdef_search_ctx->copy_fn(&in[(-yoff * CDEF_BSTRIDE - xoff)], CDEF_BSTRIDE,
602
393k
                             pd.dst.buf, row - yoff, col - xoff, pd.dst.stride,
603
393k
                             ysize, xsize);
604
393k
    fill_borders_for_fbs_on_frame_boundary(
605
393k
        inbuf, hfilt_size, vfilt_size, is_fb_on_frm_left_boundary,
606
393k
        is_fb_on_frm_right_boundary, is_fb_on_frm_top_boundary,
607
393k
        is_fb_on_frm_bottom_boundary);
608
1.96M
    for (int gi = 0; gi < cdef_search_ctx->total_strengths; gi++) {
609
1.56M
      int pri_strength, sec_strength;
610
1.56M
      get_cdef_filter_strengths(cdef_search_ctx->pick_method, &pri_strength,
611
1.56M
                                &sec_strength, gi);
612
1.56M
      const uint64_t curr_mse = get_filt_error(
613
1.56M
          cdef_search_ctx, &pd, dlist, dir, &dirinit, var, in, ref_buffer[pli],
614
1.56M
          ref_stride[pli], row, col, pri_strength, sec_strength, cdef_count,
615
1.56M
          pli, coeff_shift, bs);
616
1.56M
      if (pli < 2)
617
1.19M
        cdef_search_ctx->mse[pli][sb_count][gi] = curr_mse;
618
378k
      else
619
378k
        cdef_search_ctx->mse[1][sb_count][gi] += curr_mse;
620
1.56M
    }
621
393k
  }
622
202k
  cdef_search_ctx->sb_index[sb_count] =
623
202k
      MI_SIZE_64X64 * fbr * mi_params->mi_stride + MI_SIZE_64X64 * fbc;
624
202k
}
625
626
// MSE calculation at frame level.
627
// Inputs:
628
//   cdef_search_ctx: Pointer to the structure containing parameters related to
629
//   CDEF search context.
630
//   adaptive_cdef_mode: Speed feature to control CDEF adaptively.
631
// Returns:
632
//   Nothing will be returned. Contents of cdef_search_ctx will be modified.
633
static void cdef_mse_calc_frame(CdefSearchCtx *cdef_search_ctx,
634
                                struct aom_internal_error_info *error_info,
635
21.7k
                                int adaptive_cdef_mode) {
636
  // Loop over each sb.
637
50.3k
  for (int fbr = 0; fbr < cdef_search_ctx->nvfb; ++fbr) {
638
77.1k
    for (int fbc = 0; fbc < cdef_search_ctx->nhfb; ++fbc) {
639
      // Checks if cdef processing can be skipped for particular sb.
640
48.5k
      if (cdef_sb_skip(cdef_search_ctx->mi_params, fbr, fbc)) continue;
641
      // Calculate mse for each sb and store the relevant sb index.
642
45.8k
      av1_cdef_mse_calc_block(cdef_search_ctx, error_info, fbr, fbc,
643
45.8k
                              cdef_search_ctx->sb_count, adaptive_cdef_mode);
644
45.8k
      cdef_search_ctx->sb_count++;
645
45.8k
    }
646
28.5k
  }
647
21.7k
}
648
649
// Allocates memory for members of CdefSearchCtx.
650
// Inputs:
651
//   cdef_search_ctx: Pointer to the structure containing parameters
652
//   related to CDEF search context.
653
// Returns:
654
//   Nothing will be returned. Contents of cdef_search_ctx will be modified.
655
61.6k
static void cdef_alloc_data(AV1_COMMON *cm, CdefSearchCtx *cdef_search_ctx) {
656
61.6k
  const int nvfb = cdef_search_ctx->nvfb;
657
61.6k
  const int nhfb = cdef_search_ctx->nhfb;
658
61.6k
  CHECK_MEM_ERROR(
659
61.6k
      cm, cdef_search_ctx->sb_index,
660
61.6k
      aom_malloc(nvfb * nhfb * sizeof(cdef_search_ctx->sb_index[0])));
661
61.6k
  cdef_search_ctx->sb_count = 0;
662
61.6k
  CHECK_MEM_ERROR(cm, cdef_search_ctx->mse[0],
663
61.6k
                  aom_malloc(sizeof(**cdef_search_ctx->mse) * nvfb * nhfb));
664
61.6k
  CHECK_MEM_ERROR(cm, cdef_search_ctx->mse[1],
665
61.6k
                  aom_malloc(sizeof(**cdef_search_ctx->mse) * nvfb * nhfb));
666
61.6k
}
667
668
// Deallocates the memory allocated for members of CdefSearchCtx.
669
// Inputs:
670
//   cdef_search_ctx: Pointer to the structure containing parameters
671
//   related to CDEF search context.
672
// Returns:
673
//   Nothing will be returned.
674
156k
void av1_cdef_dealloc_data(CdefSearchCtx *cdef_search_ctx) {
675
156k
  if (cdef_search_ctx) {
676
100k
    aom_free(cdef_search_ctx->mse[0]);
677
100k
    cdef_search_ctx->mse[0] = NULL;
678
100k
    aom_free(cdef_search_ctx->mse[1]);
679
100k
    cdef_search_ctx->mse[1] = NULL;
680
100k
    aom_free(cdef_search_ctx->sb_index);
681
100k
    cdef_search_ctx->sb_index = NULL;
682
100k
  }
683
156k
}
684
685
// Initialize the parameters related to CDEF search context.
686
// Inputs:
687
//   frame: Pointer to compressed frame buffer
688
//   ref: Pointer to the frame buffer holding the source frame
689
//   cm: Pointer to top level common structure
690
//   xd: Pointer to common current coding block structure
691
//   cdef_search_ctx: Pointer to the structure containing parameters related to
692
//   CDEF search context.
693
//   pick_method: Search method used to select CDEF parameters
694
// Returns:
695
//   Nothing will be returned. Contents of cdef_search_ctx will be modified.
696
static inline void cdef_params_init(const YV12_BUFFER_CONFIG *frame,
697
                                    const YV12_BUFFER_CONFIG *ref,
698
                                    AV1_COMMON *cm, MACROBLOCKD *xd,
699
                                    CdefSearchCtx *cdef_search_ctx,
700
61.6k
                                    CDEF_PICK_METHOD pick_method) {
701
61.6k
  const CommonModeInfoParams *const mi_params = &cm->mi_params;
702
61.6k
  const int num_planes = av1_num_planes(cm);
703
61.6k
  cdef_search_ctx->mi_params = &cm->mi_params;
704
61.6k
  cdef_search_ctx->ref = ref;
705
61.6k
  cdef_search_ctx->nvfb =
706
61.6k
      (mi_params->mi_rows + MI_SIZE_64X64 - 1) / MI_SIZE_64X64;
707
61.6k
  cdef_search_ctx->nhfb =
708
61.6k
      (mi_params->mi_cols + MI_SIZE_64X64 - 1) / MI_SIZE_64X64;
709
61.6k
  cdef_search_ctx->coeff_shift = AOMMAX(cm->seq_params->bit_depth - 8, 0);
710
61.6k
  cdef_search_ctx->damping = 3 + (cm->quant_params.base_qindex >> 6);
711
61.6k
  cdef_search_ctx->total_strengths = nb_cdef_strengths[pick_method];
712
61.6k
  cdef_search_ctx->num_planes = num_planes;
713
61.6k
  cdef_search_ctx->pick_method = pick_method;
714
61.6k
  cdef_search_ctx->sb_count = 0;
715
61.6k
  cdef_search_ctx->use_highbitdepth = cm->seq_params->use_highbitdepth;
716
61.6k
  av1_setup_dst_planes(xd->plane, cm->seq_params->sb_size, frame, 0, 0, 0,
717
61.6k
                       num_planes);
718
  // Initialize plane wise information.
719
177k
  for (int pli = 0; pli < num_planes; pli++) {
720
115k
    cdef_search_ctx->xdec[pli] = xd->plane[pli].subsampling_x;
721
115k
    cdef_search_ctx->ydec[pli] = xd->plane[pli].subsampling_y;
722
115k
    cdef_search_ctx->bsize[pli] =
723
115k
        cdef_search_ctx->ydec[pli]
724
115k
            ? (cdef_search_ctx->xdec[pli] ? BLOCK_4X4 : BLOCK_8X4)
725
115k
            : (cdef_search_ctx->xdec[pli] ? BLOCK_4X8 : BLOCK_8X8);
726
115k
    cdef_search_ctx->mi_wide_l2[pli] =
727
115k
        MI_SIZE_LOG2 - xd->plane[pli].subsampling_x;
728
115k
    cdef_search_ctx->mi_high_l2[pli] =
729
115k
        MI_SIZE_LOG2 - xd->plane[pli].subsampling_y;
730
115k
    cdef_search_ctx->plane[pli] = xd->plane[pli];
731
115k
  }
732
  // Function pointer initialization.
733
61.6k
#if CONFIG_AV1_HIGHBITDEPTH
734
61.6k
  if (cm->seq_params->use_highbitdepth) {
735
19.7k
    cdef_search_ctx->copy_fn = av1_cdef_copy_sb8_16_highbd;
736
19.7k
    cdef_search_ctx->compute_cdef_dist_fn = compute_cdef_dist_highbd;
737
41.8k
  } else {
738
41.8k
    cdef_search_ctx->copy_fn = av1_cdef_copy_sb8_16_lowbd;
739
41.8k
    cdef_search_ctx->compute_cdef_dist_fn = compute_cdef_dist;
740
41.8k
  }
741
#else
742
  cdef_search_ctx->copy_fn = av1_cdef_copy_sb8_16_lowbd;
743
  cdef_search_ctx->compute_cdef_dist_fn = compute_cdef_dist;
744
#endif
745
61.6k
}
746
747
void av1_pick_cdef_from_qp(AV1_COMMON *const cm, int skip_cdef,
748
37.2k
                           int is_screen_content, bool avoid_uv_cdef) {
749
37.2k
  const int bd = cm->seq_params->bit_depth;
750
37.2k
  const int q =
751
37.2k
      av1_ac_quant_QTX(cm->quant_params.base_qindex, 0, bd) >> (bd - 8);
752
37.2k
  CdefInfo *const cdef_info = &cm->cdef_info;
753
  // Check the speed feature to avoid extra signaling.
754
37.2k
  if (skip_cdef) {
755
0
    cdef_info->cdef_bits = 1;
756
0
    cdef_info->nb_cdef_strengths = 2;
757
37.2k
  } else {
758
37.2k
    cdef_info->cdef_bits = 0;
759
37.2k
    cdef_info->nb_cdef_strengths = 1;
760
37.2k
  }
761
37.2k
  cdef_info->cdef_damping = 3 + (cm->quant_params.base_qindex >> 6);
762
763
37.2k
  int predicted_y_f1 = 0;
764
37.2k
  int predicted_y_f2 = 0;
765
37.2k
  int predicted_uv_f1 = 0;
766
37.2k
  int predicted_uv_f2 = 0;
767
37.2k
  if (is_screen_content) {
768
0
    predicted_y_f1 =
769
0
        (int)(5.88217781e-06 * q * q + 6.10391455e-03 * q + 9.95043102e-02);
770
0
    predicted_y_f2 =
771
0
        (int)(-7.79934857e-06 * q * q + 6.58957830e-03 * q + 8.81045025e-01);
772
0
    predicted_uv_f1 =
773
0
        (int)(-6.79500136e-06 * q * q + 1.02695586e-02 * q + 1.36126802e-01);
774
0
    predicted_uv_f2 =
775
0
        (int)(-9.99613695e-08 * q * q - 1.79361339e-05 * q + 1.17022324e+0);
776
0
    predicted_y_f1 = clamp(predicted_y_f1, 0, 15);
777
0
    predicted_y_f2 = clamp(predicted_y_f2, 0, 3);
778
0
    predicted_uv_f1 = clamp(predicted_uv_f1, 0, 15);
779
0
    predicted_uv_f2 = clamp(predicted_uv_f2, 0, 3);
780
37.2k
  } else {
781
37.2k
    if (!frame_is_intra_only(cm)) {
782
9.24k
      predicted_y_f1 = clamp((int)roundf(q * q * -0.0000023593946f +
783
9.24k
                                         q * 0.0068615186f + 0.02709886f),
784
9.24k
                             0, 15);
785
9.24k
      predicted_y_f2 = clamp((int)roundf(q * q * -0.00000057629734f +
786
9.24k
                                         q * 0.0013993345f + 0.03831067f),
787
9.24k
                             0, 3);
788
9.24k
      predicted_uv_f1 = clamp((int)roundf(q * q * -0.0000007095069f +
789
9.24k
                                          q * 0.0034628846f + 0.00887099f),
790
9.24k
                              0, 15);
791
9.24k
      predicted_uv_f2 = clamp((int)roundf(q * q * 0.00000023874085f +
792
9.24k
                                          q * 0.00028223585f + 0.05576307f),
793
9.24k
                              0, 3);
794
28.0k
    } else {
795
28.0k
      predicted_y_f1 = clamp(
796
28.0k
          (int)roundf(q * q * 0.0000033731974f + q * 0.008070594f + 0.0187634f),
797
28.0k
          0, 15);
798
28.0k
      predicted_y_f2 = clamp((int)roundf(q * q * 0.0000029167343f +
799
28.0k
                                         q * 0.0027798624f + 0.0079405f),
800
28.0k
                             0, 3);
801
28.0k
      predicted_uv_f1 = clamp((int)roundf(q * q * -0.0000130790995f +
802
28.0k
                                          q * 0.012892405f - 0.00748388f),
803
28.0k
                              0, 15);
804
28.0k
      predicted_uv_f2 = clamp((int)roundf(q * q * 0.0000032651783f +
805
28.0k
                                          q * 0.00035520183f + 0.00228092f),
806
28.0k
                              0, 3);
807
28.0k
    }
808
37.2k
  }
809
37.2k
  cdef_info->cdef_strengths[0] =
810
37.2k
      predicted_y_f1 * CDEF_SEC_STRENGTHS + predicted_y_f2;
811
37.2k
  cdef_info->cdef_uv_strengths[0] =
812
37.2k
      avoid_uv_cdef ? 0
813
37.2k
                    : predicted_uv_f1 * CDEF_SEC_STRENGTHS + predicted_uv_f2;
814
815
  // mbmi->cdef_strength is already set in the encoding stage. We don't need to
816
  // set it again here.
817
37.2k
  if (skip_cdef) {
818
0
    cdef_info->cdef_strengths[1] = 0;
819
0
    cdef_info->cdef_uv_strengths[1] = 0;
820
0
    return;
821
0
  }
822
823
37.2k
  const CommonModeInfoParams *const mi_params = &cm->mi_params;
824
37.2k
  const int nvfb = (mi_params->mi_rows + MI_SIZE_64X64 - 1) / MI_SIZE_64X64;
825
37.2k
  const int nhfb = (mi_params->mi_cols + MI_SIZE_64X64 - 1) / MI_SIZE_64X64;
826
37.2k
  MB_MODE_INFO **mbmi = mi_params->mi_grid_base;
827
  // mbmi is NULL when real-time rate control library is used.
828
37.2k
  if (!mbmi) return;
829
106k
  for (int r = 0; r < nvfb; ++r) {
830
210k
    for (int c = 0; c < nhfb; ++c) {
831
141k
      MB_MODE_INFO *current_mbmi = mbmi[MI_SIZE_64X64 * c];
832
141k
      current_mbmi->cdef_strength = 0;
833
141k
    }
834
69.3k
    mbmi += MI_SIZE_64X64 * mi_params->mi_stride;
835
69.3k
  }
836
37.2k
}
837
838
104k
void av1_cdef_search(AV1_COMP *cpi) {
839
104k
  AV1_COMMON *cm = &cpi->common;
840
104k
  CDEF_CONTROL cdef_control = cpi->oxcf.tool_cfg.cdef_control;
841
104k
  const bool apply_adaptive_cdef =
842
104k
      cdef_control == CDEF_ADAPTIVE &&
843
51.0k
      (cpi->oxcf.rc_cfg.mode == AOM_Q || cpi->oxcf.rc_cfg.mode == AOM_CQ);
844
845
104k
  assert(cdef_control != CDEF_NONE);
846
  // For CDEF_ADAPTIVE, turning off CDEF around qindex 32 was best for still
847
  // pictures
848
104k
  if ((cdef_control == CDEF_REFERENCE &&
849
0
       cpi->ppi->rtc_ref.non_reference_frame) ||
850
104k
      (apply_adaptive_cdef && cpi->oxcf.rc_cfg.cq_level <= 32)) {
851
5.33k
    CdefInfo *const cdef_info = &cm->cdef_info;
852
5.33k
    cdef_info->nb_cdef_strengths = 1;
853
5.33k
    cdef_info->cdef_bits = 0;
854
5.33k
    cdef_info->cdef_strengths[0] = 0;
855
5.33k
    cdef_info->cdef_uv_strengths[0] = 0;
856
5.33k
    return;
857
5.33k
  }
858
859
  // Indicate if external RC is used for testing
860
98.9k
  const int rtc_ext_rc = cpi->rc.rtc_external_ratectrl;
861
98.9k
  if (rtc_ext_rc) {
862
0
    av1_pick_cdef_from_qp(cm, /*skip_cdef=*/0, /*is_screen_content=*/0,
863
0
                          /*avoid_uv_cdef=*/false);
864
0
    return;
865
0
  }
866
98.9k
  CDEF_PICK_METHOD pick_method = cpi->sf.lpf_sf.cdef_pick_method;
867
98.9k
  if (pick_method == CDEF_PICK_FROM_Q) {
868
37.2k
    const int use_screen_content_model =
869
37.2k
        cm->quant_params.base_qindex >
870
37.2k
            AOMMAX(cpi->sf.rt_sf.screen_content_cdef_filter_qindex_thresh,
871
37.2k
                   cpi->rc.best_quality + 5) &&
872
17.8k
        cpi->oxcf.tune_cfg.content == AOM_CONTENT_SCREEN;
873
874
    // For adaptive CDEF, do not apply CDEF to chroma channels.
875
    // This is done to reduce decode time, as CDEF is a relatively-expensive
876
    // filter to compute.
877
37.2k
    const bool avoid_uv_cdef = apply_adaptive_cdef;
878
879
37.2k
    av1_pick_cdef_from_qp(cm, cpi->sf.rt_sf.skip_cdef_sb != 0,
880
37.2k
                          use_screen_content_model, avoid_uv_cdef);
881
37.2k
    return;
882
37.2k
  }
883
61.6k
  const CommonModeInfoParams *const mi_params = &cm->mi_params;
884
61.6k
  const int damping = 3 + (cm->quant_params.base_qindex >> 6);
885
61.6k
  const int fast = (pick_method >= CDEF_FAST_SEARCH_LVL1 &&
886
61.6k
                    pick_method <= CDEF_FAST_SEARCH_LVL5);
887
61.6k
  const int adaptive_cdef_mode = cpi->sf.lpf_sf.adaptive_cdef_mode;
888
61.6k
  const int num_planes = av1_num_planes(cm);
889
61.6k
  MACROBLOCKD *xd = &cpi->td.mb.e_mbd;
890
891
61.6k
  if (!cpi->cdef_search_ctx)
892
61.6k
    CHECK_MEM_ERROR(cm, cpi->cdef_search_ctx,
893
61.6k
                    aom_calloc(1, sizeof(*cpi->cdef_search_ctx)));
894
61.6k
  CdefSearchCtx *cdef_search_ctx = cpi->cdef_search_ctx;
895
896
  // Initialize parameters related to CDEF search context.
897
61.6k
  cdef_params_init(&cm->cur_frame->buf, cpi->source, cm, xd, cdef_search_ctx,
898
61.6k
                   pick_method);
899
  // Allocate CDEF search context buffers.
900
61.6k
  cdef_alloc_data(cm, cdef_search_ctx);
901
  // Frame level mse calculation.
902
61.6k
  if (cpi->mt_info.num_workers > 1) {
903
39.8k
    av1_cdef_mse_calc_frame_mt(cpi);
904
39.8k
  } else {
905
21.7k
    cdef_mse_calc_frame(cdef_search_ctx, cm->error, adaptive_cdef_mode);
906
21.7k
  }
907
908
  /* Search for different number of signaling bits. */
909
61.6k
  int nb_strength_bits = 0;
910
61.6k
  uint64_t best_rd = UINT64_MAX;
911
61.6k
  CdefInfo *const cdef_info = &cm->cdef_info;
912
61.6k
  int sb_count = cdef_search_ctx->sb_count;
913
61.6k
  uint64_t(*mse[2])[TOTAL_STRENGTHS];
914
61.6k
  mse[0] = cdef_search_ctx->mse[0];
915
61.6k
  mse[1] = cdef_search_ctx->mse[1];
916
  /* Calculate the maximum number of bits required to signal CDEF strengths at
917
   * block level */
918
61.6k
  const int total_strengths = nb_cdef_strengths[pick_method];
919
61.6k
  const int joint_strengths =
920
61.6k
      num_planes > 1 ? total_strengths * total_strengths : total_strengths;
921
61.6k
  const int max_signaling_bits =
922
61.6k
      joint_strengths == 1 ? 0 : get_msb(joint_strengths - 1) + 1;
923
61.6k
  int rdmult = cpi->td.mb.rdmult;
924
925
  // For adaptive CDEF, reduce primary and secondary CDEF strengths for
926
  // qindexes up to 220.
927
61.6k
  const bool should_reduce_cdef_strengths =
928
61.6k
      apply_adaptive_cdef && cpi->oxcf.rc_cfg.cq_level <= 220;
929
  // For adaptive CDEF with strength reduction, zero out CDEF strengths with
930
  // low values (luma and/or chroma). This is done to reduce decode time, as
931
  // CDEF is a relatively-expensive filter to compute.
932
61.6k
  const bool should_zero_cdef_strengths =
933
61.6k
      should_reduce_cdef_strengths && cpi->sf.lpf_sf.zero_low_cdef_strengths;
934
  // If running adaptive CDEF with strength zeroing, let search derive at least
935
  // two CDEF strengths (i.e. at least 1 CDEF signaling bit), unless search was
936
  // explicitly set to search for 1 strength only (i.e. 0 CDEF signaling bits).
937
  // Doing so will help find opportunities to zero out low strengths to reduce
938
  // overall decode time.
939
61.6k
  const int min_signaling_bits =
940
61.6k
      (should_zero_cdef_strengths && max_signaling_bits > 0) ? 1 : 0;
941
942
264k
  for (int i = min_signaling_bits; i <= 3; i++) {
943
237k
    if (i > max_signaling_bits) break;
944
203k
    int best_lev0[CDEF_MAX_STRENGTHS] = { 0 };
945
203k
    int best_lev1[CDEF_MAX_STRENGTHS] = { 0 };
946
203k
    const int nb_strengths = 1 << i;
947
203k
    uint64_t tot_mse;
948
203k
    if (num_planes > 1) {
949
103k
      tot_mse = joint_strength_search_dual(best_lev0, best_lev1, nb_strengths,
950
103k
                                           mse, sb_count, pick_method);
951
103k
    } else {
952
99.4k
      tot_mse = joint_strength_search(best_lev0, nb_strengths, mse[0], sb_count,
953
99.4k
                                      pick_method);
954
99.4k
    }
955
956
203k
    const int total_bits = sb_count * i + nb_strengths * CDEF_STRENGTH_BITS *
957
203k
                                              (num_planes > 1 ? 2 : 1);
958
203k
    const int rate_cost = av1_cost_literal(total_bits);
959
203k
    const uint64_t dist = tot_mse * 16;
960
203k
    const uint64_t rd = RDCOST(rdmult, rate_cost, dist);
961
203k
    if (rd < best_rd) {
962
71.2k
      best_rd = rd;
963
71.2k
      nb_strength_bits = i;
964
71.2k
      memcpy(cdef_info->cdef_strengths, best_lev0,
965
71.2k
             nb_strengths * sizeof(best_lev0[0]));
966
71.2k
      if (num_planes > 1) {
967
31.4k
        memcpy(cdef_info->cdef_uv_strengths, best_lev1,
968
31.4k
               nb_strengths * sizeof(best_lev1[0]));
969
31.4k
      }
970
71.2k
    }
971
203k
  }
972
973
61.6k
  cdef_info->cdef_bits = nb_strength_bits;
974
61.6k
  cdef_info->nb_cdef_strengths = 1 << nb_strength_bits;
975
61.6k
  uint64_t tot_best_mse_y = 0, tot_zero_mse_y = 0;
976
264k
  for (int i = 0; i < sb_count; i++) {
977
202k
    uint64_t best_mse = UINT64_MAX;
978
202k
    int best_gi = 0;
979
202k
    tot_zero_mse_y += mse[0][i][0];
980
544k
    for (int gi = 0; gi < cdef_info->nb_cdef_strengths; gi++) {
981
342k
      uint64_t curr = mse[0][i][cdef_info->cdef_strengths[gi]];
982
342k
      if (num_planes > 1) curr += mse[1][i][cdef_info->cdef_uv_strengths[gi]];
983
342k
      if (curr < best_mse) {
984
240k
        best_gi = gi;
985
240k
        best_mse = curr;
986
240k
      }
987
342k
    }
988
    // When adaptive_cdef_mode is enabled, CDEF filter of chroma is disabled by
989
    // setting the MSE at chroma strength index 0 to zero, ensuring zero filter
990
    // strength always wins for chroma. Hence, best_mse reflects luma MSE only.
991
202k
    tot_best_mse_y += best_mse;
992
202k
    mi_params->mi_grid_base[cdef_search_ctx->sb_index[i]]->cdef_strength =
993
202k
        best_gi;
994
202k
  }
995
996
  // Disable CDEF filter of luma if the MSE improvement over zero filter
997
  // strength is below the threshold.
998
61.6k
  if (cm->current_frame.pyramid_level > 1 &&
999
11.9k
      cpi->sf.lpf_sf.adaptive_cdef_mode > 0 && sb_count > 0) {
1000
0
    const double cdef_mse_pct_imp_thresh = 4.0;
1001
0
    double luma_mse_gain_pct = 0.0;
1002
    // Percentage reduction in luma MSE with best CDEF strength vs zero
1003
    // strength.
1004
0
    luma_mse_gain_pct =
1005
0
        (tot_zero_mse_y - tot_best_mse_y) * 100.0 / tot_zero_mse_y;
1006
0
    if (luma_mse_gain_pct < cdef_mse_pct_imp_thresh) {
1007
0
      cdef_info->nb_cdef_strengths = 1;
1008
0
      cdef_info->cdef_bits = 0;
1009
0
      cdef_info->cdef_strengths[0] = 0;
1010
0
      cdef_info->cdef_uv_strengths[0] = 0;
1011
0
      for (int i = 0; i < sb_count; i++) {
1012
0
        mi_params->mi_grid_base[cdef_search_ctx->sb_index[i]]->cdef_strength =
1013
0
            0;
1014
0
      }
1015
0
    }
1016
0
  }
1017
1018
61.6k
  if (fast) {
1019
143k
    for (int j = 0; j < cdef_info->nb_cdef_strengths; j++) {
1020
82.0k
      const int luma_strength = cdef_info->cdef_strengths[j];
1021
82.0k
      int pri_strength, sec_strength;
1022
1023
82.0k
      STORE_CDEF_FILTER_STRENGTH(cdef_info->cdef_strengths[j], pick_method,
1024
82.0k
                                 luma_strength);
1025
82.0k
      if (num_planes > 1) {
1026
36.9k
        const int chroma_strength = cdef_info->cdef_uv_strengths[j];
1027
36.9k
        STORE_CDEF_FILTER_STRENGTH(cdef_info->cdef_uv_strengths[j], pick_method,
1028
36.9k
                                   chroma_strength);
1029
36.9k
      }
1030
82.0k
    }
1031
61.6k
  }
1032
1033
  // Perform CDEF strength reduction.
1034
  // Note 1: for odd strengths, the 0.5 discarded by ">> 1" is a significant
1035
  // part of the strength when the strength is small, and because there are
1036
  // few strength levels, odd strengths are reduced significantly more than a
1037
  // half. This is intended behavior for reduced strength.
1038
  // For example: a pri strength of 3 becomes 1, and a sec strength of 1
1039
  // becomes 0.
1040
  // Note 2: a (signaled) sec strength value of 3 is special as it results in an
1041
  // actual sec strength of 4. We tried adding +1 to the sec strength 3 so it
1042
  // maps to a reduced sec strength of 2. However, on Daala's subset1, the
1043
  // resulting SSIMULACRA 2 scores were either exactly the same (at cpu-used 6),
1044
  // or within noise level (at cpu-used 3). Given that there were no discernible
1045
  // improvements, this special mapping was left out for reduced strength.
1046
61.6k
  if (should_reduce_cdef_strengths) {
1047
51.9k
    for (int j = 0; j < cdef_info->nb_cdef_strengths; j++) {
1048
31.5k
      const int luma_strength = cdef_info->cdef_strengths[j];
1049
31.5k
      const int new_pri_luma_strength =
1050
31.5k
          (luma_strength / CDEF_SEC_STRENGTHS) >> 1;
1051
31.5k
      const int new_sec_luma_strength =
1052
31.5k
          (luma_strength % CDEF_SEC_STRENGTHS) >> 1;
1053
1054
31.5k
      cdef_info->cdef_strengths[j] =
1055
31.5k
          new_pri_luma_strength * CDEF_SEC_STRENGTHS + new_sec_luma_strength;
1056
1057
31.5k
      int new_pri_chroma_strength;
1058
31.5k
      int new_sec_chroma_strength;
1059
1060
31.5k
      if (num_planes > 1) {
1061
15.6k
        const int chroma_strength = cdef_info->cdef_uv_strengths[j];
1062
15.6k
        new_pri_chroma_strength = (chroma_strength / CDEF_SEC_STRENGTHS) >> 1;
1063
15.6k
        new_sec_chroma_strength = (chroma_strength % CDEF_SEC_STRENGTHS) >> 1;
1064
1065
15.6k
        cdef_info->cdef_uv_strengths[j] =
1066
15.6k
            new_pri_chroma_strength * CDEF_SEC_STRENGTHS +
1067
15.6k
            new_sec_chroma_strength;
1068
15.6k
      }
1069
1070
      // Zero out entries with low CDEF luma (and optional chroma) strengths.
1071
      // The low-strength thresholds were empirically derived from subjective
1072
      // testing and SSIMULACRA 2 scores. These strike a balance between
1073
      // perceptual quality gains and a reasonable single-threaded decode time
1074
      // increase (~10%) over --enable-cdef 0. There's an overall 0.18 point
1075
      // loss in SSIMULACRA 2 scores over no CDEF strength zeroing at speed 6,
1076
      // QP 30 on the CLIC 2020 dataset.
1077
31.5k
      if (should_zero_cdef_strengths) {
1078
17.6k
        const bool is_low_luma_strength =
1079
17.6k
            new_pri_luma_strength <= 4 && new_sec_luma_strength <= 1;
1080
1081
17.6k
        if (is_low_luma_strength) {
1082
10.7k
          cdef_info->cdef_strengths[j] = 0;
1083
10.7k
        }
1084
17.6k
        if (num_planes > 1) {
1085
8.64k
          const bool is_low_chroma_strength =
1086
8.64k
              new_pri_chroma_strength <= 4 && new_sec_chroma_strength <= 1;
1087
1088
8.64k
          if (is_low_luma_strength || is_low_chroma_strength) {
1089
            // Disable CDEF on chroma if we've disabled it on luma
1090
6.03k
            cdef_info->cdef_uv_strengths[j] = 0;
1091
6.03k
          }
1092
8.64k
        }
1093
17.6k
      }
1094
31.5k
    }
1095
20.4k
  }
1096
1097
61.6k
  cdef_info->cdef_damping = damping;
1098
  // Deallocate CDEF search context buffers.
1099
61.6k
  av1_cdef_dealloc_data(cdef_search_ctx);
1100
61.6k
}