Coverage Report

Created: 2026-09-07 06:44

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/aom/aom_dsp/x86/adaptive_quantize_sse2.c
Line
Count
Source
1
/*
2
 * Copyright (c) 2019, Alliance for Open Media. All rights reserved.
3
 *
4
 * This source code is subject to the terms of the BSD 2 Clause License and
5
 * the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
6
 * was not distributed with this source code in the LICENSE file, you can
7
 * obtain it at www.aomedia.org/license/software. If the Alliance for Open
8
 * Media Patent License 1.0 was not distributed with this source code in the
9
 * PATENTS file, you can obtain it at www.aomedia.org/license/patent.
10
 */
11
12
#include <assert.h>
13
#include <emmintrin.h>
14
#include "config/aom_dsp_rtcd.h"
15
#include "aom/aom_integer.h"
16
#include "aom_dsp/quantize.h"
17
#include "aom_dsp/x86/quantize_x86.h"
18
19
void aom_quantize_b_adaptive_sse2(
20
    const tran_low_t *coeff_ptr, intptr_t n_coeffs, const int16_t *zbin_ptr,
21
    const int16_t *round_ptr, const int16_t *quant_ptr,
22
    const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr,
23
    tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr,
24
0
    const int16_t *scan, const int16_t *iscan) {
25
0
  int index = 16;
26
0
  int non_zero_count = 0;
27
0
  int non_zero_count_prescan_add_zero = 0;
28
0
  int is_found0 = 0, is_found1 = 0;
29
0
  int eob = -1;
30
0
  const __m128i zero = _mm_setzero_si128();
31
0
  __m128i zbin, round, quant, dequant, shift;
32
0
  __m128i coeff0, coeff1, coeff0_sign, coeff1_sign;
33
0
  __m128i qcoeff0, qcoeff1;
34
0
  __m128i cmp_mask0, cmp_mask1;
35
0
  __m128i all_zero;
36
0
  __m128i mask0 = zero, mask1 = zero;
37
38
0
  int prescan_add[2];
39
0
  int thresh[4];
40
0
  const qm_val_t wt = (1 << AOM_QM_BITS);
41
0
  for (int i = 0; i < 2; ++i) {
42
0
    prescan_add[i] = ROUND_POWER_OF_TWO(dequant_ptr[i] * EOB_FACTOR, 7);
43
0
    thresh[i] = (zbin_ptr[i] * wt + prescan_add[i]) - 1;
44
0
  }
45
0
  thresh[2] = thresh[3] = thresh[1];
46
0
  __m128i threshold[2];
47
0
  threshold[0] = _mm_loadu_si128((__m128i *)&thresh[0]);
48
0
  threshold[1] = _mm_unpackhi_epi64(threshold[0], threshold[0]);
49
50
0
#if SKIP_EOB_FACTOR_ADJUST
51
0
  int first = -1;
52
0
#endif
53
  // Setup global values.
54
0
  load_b_values(zbin_ptr, &zbin, round_ptr, &round, quant_ptr, &quant,
55
0
                dequant_ptr, &dequant, quant_shift_ptr, &shift);
56
57
  // Do DC and first 15 AC.
58
0
  coeff0 = load_coefficients(coeff_ptr);
59
0
  coeff1 = load_coefficients(coeff_ptr + 8);
60
61
  // Poor man's abs().
62
0
  coeff0_sign = _mm_srai_epi16(coeff0, 15);
63
0
  coeff1_sign = _mm_srai_epi16(coeff1, 15);
64
0
  qcoeff0 = invert_sign_sse2(coeff0, coeff0_sign);
65
0
  qcoeff1 = invert_sign_sse2(coeff1, coeff1_sign);
66
67
0
  update_mask0(&qcoeff0, &qcoeff1, threshold, iscan, &is_found0, &mask0);
68
69
0
  cmp_mask0 = _mm_cmpgt_epi16(qcoeff0, zbin);
70
0
  zbin = _mm_unpackhi_epi64(zbin, zbin);  // Switch DC to AC
71
0
  cmp_mask1 = _mm_cmpgt_epi16(qcoeff1, zbin);
72
73
0
  update_mask1(&cmp_mask0, &cmp_mask1, iscan, &is_found1, &mask1);
74
75
0
  threshold[0] = threshold[1];
76
0
  all_zero = _mm_or_si128(cmp_mask0, cmp_mask1);
77
0
  if (_mm_movemask_epi8(all_zero) == 0) {
78
0
    _mm_store_si128((__m128i *)(qcoeff_ptr), zero);
79
0
    _mm_store_si128((__m128i *)(qcoeff_ptr + 4), zero);
80
0
    _mm_store_si128((__m128i *)(qcoeff_ptr + 8), zero);
81
0
    _mm_store_si128((__m128i *)(qcoeff_ptr + 12), zero);
82
0
    _mm_store_si128((__m128i *)(dqcoeff_ptr), zero);
83
0
    _mm_store_si128((__m128i *)(dqcoeff_ptr + 4), zero);
84
0
    _mm_store_si128((__m128i *)(dqcoeff_ptr + 8), zero);
85
0
    _mm_store_si128((__m128i *)(dqcoeff_ptr + 12), zero);
86
0
    round = _mm_unpackhi_epi64(round, round);
87
0
    quant = _mm_unpackhi_epi64(quant, quant);
88
0
    shift = _mm_unpackhi_epi64(shift, shift);
89
0
    dequant = _mm_unpackhi_epi64(dequant, dequant);
90
0
  } else {
91
0
    calculate_qcoeff(&qcoeff0, round, quant, shift);
92
93
0
    round = _mm_unpackhi_epi64(round, round);
94
0
    quant = _mm_unpackhi_epi64(quant, quant);
95
0
    shift = _mm_unpackhi_epi64(shift, shift);
96
97
0
    calculate_qcoeff(&qcoeff1, round, quant, shift);
98
99
    // Reinsert signs
100
0
    qcoeff0 = invert_sign_sse2(qcoeff0, coeff0_sign);
101
0
    qcoeff1 = invert_sign_sse2(qcoeff1, coeff1_sign);
102
103
    // Mask out zbin threshold coeffs
104
0
    qcoeff0 = _mm_and_si128(qcoeff0, cmp_mask0);
105
0
    qcoeff1 = _mm_and_si128(qcoeff1, cmp_mask1);
106
107
0
    store_coefficients(qcoeff0, qcoeff_ptr);
108
0
    store_coefficients(qcoeff1, qcoeff_ptr + 8);
109
110
0
    coeff0 = calculate_dqcoeff(qcoeff0, dequant);
111
0
    dequant = _mm_unpackhi_epi64(dequant, dequant);
112
0
    coeff1 = calculate_dqcoeff(qcoeff1, dequant);
113
114
0
    store_coefficients(coeff0, dqcoeff_ptr);
115
0
    store_coefficients(coeff1, dqcoeff_ptr + 8);
116
0
  }
117
118
  // AC only loop.
119
0
  while (index < n_coeffs) {
120
0
    coeff0 = load_coefficients(coeff_ptr + index);
121
0
    coeff1 = load_coefficients(coeff_ptr + index + 8);
122
123
0
    coeff0_sign = _mm_srai_epi16(coeff0, 15);
124
0
    coeff1_sign = _mm_srai_epi16(coeff1, 15);
125
0
    qcoeff0 = invert_sign_sse2(coeff0, coeff0_sign);
126
0
    qcoeff1 = invert_sign_sse2(coeff1, coeff1_sign);
127
128
0
    update_mask0(&qcoeff0, &qcoeff1, threshold, iscan + index, &is_found0,
129
0
                 &mask0);
130
131
0
    cmp_mask0 = _mm_cmpgt_epi16(qcoeff0, zbin);
132
0
    cmp_mask1 = _mm_cmpgt_epi16(qcoeff1, zbin);
133
134
0
    update_mask1(&cmp_mask0, &cmp_mask1, iscan + index, &is_found1, &mask1);
135
136
0
    all_zero = _mm_or_si128(cmp_mask0, cmp_mask1);
137
0
    if (_mm_movemask_epi8(all_zero) == 0) {
138
0
      _mm_store_si128((__m128i *)(qcoeff_ptr + index), zero);
139
0
      _mm_store_si128((__m128i *)(qcoeff_ptr + index + 4), zero);
140
0
      _mm_store_si128((__m128i *)(qcoeff_ptr + index + 8), zero);
141
0
      _mm_store_si128((__m128i *)(qcoeff_ptr + index + 12), zero);
142
0
      _mm_store_si128((__m128i *)(dqcoeff_ptr + index), zero);
143
0
      _mm_store_si128((__m128i *)(dqcoeff_ptr + index + 4), zero);
144
0
      _mm_store_si128((__m128i *)(dqcoeff_ptr + index + 8), zero);
145
0
      _mm_store_si128((__m128i *)(dqcoeff_ptr + index + 12), zero);
146
0
      index += 16;
147
0
      continue;
148
0
    }
149
0
    calculate_qcoeff(&qcoeff0, round, quant, shift);
150
0
    calculate_qcoeff(&qcoeff1, round, quant, shift);
151
152
0
    qcoeff0 = invert_sign_sse2(qcoeff0, coeff0_sign);
153
0
    qcoeff1 = invert_sign_sse2(qcoeff1, coeff1_sign);
154
155
0
    qcoeff0 = _mm_and_si128(qcoeff0, cmp_mask0);
156
0
    qcoeff1 = _mm_and_si128(qcoeff1, cmp_mask1);
157
158
0
    store_coefficients(qcoeff0, qcoeff_ptr + index);
159
0
    store_coefficients(qcoeff1, qcoeff_ptr + index + 8);
160
161
0
    coeff0 = calculate_dqcoeff(qcoeff0, dequant);
162
0
    coeff1 = calculate_dqcoeff(qcoeff1, dequant);
163
164
0
    store_coefficients(coeff0, dqcoeff_ptr + index);
165
0
    store_coefficients(coeff1, dqcoeff_ptr + index + 8);
166
167
0
    index += 16;
168
0
  }
169
0
  if (is_found0) non_zero_count = calculate_non_zero_count(mask0);
170
0
  if (is_found1)
171
0
    non_zero_count_prescan_add_zero = calculate_non_zero_count(mask1);
172
173
0
  for (int i = non_zero_count_prescan_add_zero - 1; i >= non_zero_count; i--) {
174
0
    const int rc = scan[i];
175
0
    qcoeff_ptr[rc] = 0;
176
0
    dqcoeff_ptr[rc] = 0;
177
0
  }
178
179
0
  for (int i = non_zero_count - 1; i >= 0; i--) {
180
0
    const int rc = scan[i];
181
0
    if (qcoeff_ptr[rc]) {
182
0
      eob = i;
183
0
      break;
184
0
    }
185
0
  }
186
187
0
  *eob_ptr = eob + 1;
188
0
#if SKIP_EOB_FACTOR_ADJUST
189
  // TODO(Aniket): Experiment the following loop with intrinsic by combining
190
  // with the quantization loop above
191
0
  for (int i = 0; i < non_zero_count; i++) {
192
0
    const int rc = scan[i];
193
0
    const int qcoeff = qcoeff_ptr[rc];
194
0
    if (qcoeff) {
195
0
      first = i;
196
0
      break;
197
0
    }
198
0
  }
199
0
  if ((*eob_ptr - 1) >= 0 && first == (*eob_ptr - 1)) {
200
0
    const int rc = scan[(*eob_ptr - 1)];
201
0
    if (qcoeff_ptr[rc] == 1 || qcoeff_ptr[rc] == -1) {
202
0
      const int coeff = coeff_ptr[rc] * wt;
203
0
      const int coeff_sign = AOMSIGN(coeff);
204
0
      const int abs_coeff = (coeff ^ coeff_sign) - coeff_sign;
205
0
      const int factor = EOB_FACTOR + SKIP_EOB_FACTOR_ADJUST;
206
0
      const int prescan_add_val =
207
0
          ROUND_POWER_OF_TWO(dequant_ptr[rc != 0] * factor, 7);
208
0
      if (abs_coeff <
209
0
          (zbin_ptr[rc != 0] * (1 << AOM_QM_BITS) + prescan_add_val)) {
210
0
        qcoeff_ptr[rc] = 0;
211
0
        dqcoeff_ptr[rc] = 0;
212
0
        *eob_ptr = 0;
213
0
      }
214
0
    }
215
0
  }
216
0
#endif
217
0
}
218
219
void aom_quantize_b_32x32_adaptive_sse2(
220
    const tran_low_t *coeff_ptr, intptr_t n_coeffs, const int16_t *zbin_ptr,
221
    const int16_t *round_ptr, const int16_t *quant_ptr,
222
    const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr,
223
    tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr,
224
0
    const int16_t *scan, const int16_t *iscan) {
225
0
  int index = 16;
226
0
  const int log_scale = 1;
227
0
  int non_zero_count = 0;
228
0
  int non_zero_count_prescan_add_zero = 0;
229
0
  int is_found0 = 0, is_found1 = 0;
230
0
  int eob = -1;
231
0
  const __m128i zero = _mm_setzero_si128();
232
0
  const __m128i one = _mm_set1_epi16(1);
233
0
  const __m128i log_scale_vec = _mm_set1_epi16(log_scale);
234
0
  __m128i zbin, round, quant, dequant, shift;
235
0
  __m128i coeff0, coeff1, coeff0_sign, coeff1_sign;
236
0
  __m128i qcoeff0, qcoeff1;
237
0
  __m128i cmp_mask0, cmp_mask1;
238
0
  __m128i all_zero;
239
0
  __m128i mask0 = zero, mask1 = zero;
240
241
0
  const int zbins[2] = { ROUND_POWER_OF_TWO(zbin_ptr[0], log_scale),
242
0
                         ROUND_POWER_OF_TWO(zbin_ptr[1], log_scale) };
243
0
  int prescan_add[2];
244
0
  int thresh[4];
245
0
  const qm_val_t wt = (1 << AOM_QM_BITS);
246
0
  for (int i = 0; i < 2; ++i) {
247
0
    prescan_add[i] = ROUND_POWER_OF_TWO(dequant_ptr[i] * EOB_FACTOR, 7);
248
0
    thresh[i] = (zbins[i] * wt + prescan_add[i]) - 1;
249
0
  }
250
0
  thresh[2] = thresh[3] = thresh[1];
251
0
  __m128i threshold[2];
252
0
  threshold[0] = _mm_loadu_si128((__m128i *)&thresh[0]);
253
0
  threshold[1] = _mm_unpackhi_epi64(threshold[0], threshold[0]);
254
255
0
#if SKIP_EOB_FACTOR_ADJUST
256
0
  int first = -1;
257
0
#endif
258
  // Setup global values.
259
0
  zbin = _mm_load_si128((const __m128i *)zbin_ptr);
260
0
  round = _mm_load_si128((const __m128i *)round_ptr);
261
0
  quant = _mm_load_si128((const __m128i *)quant_ptr);
262
0
  dequant = _mm_load_si128((const __m128i *)dequant_ptr);
263
0
  shift = _mm_load_si128((const __m128i *)quant_shift_ptr);
264
265
  // Shift with rounding.
266
0
  zbin = _mm_add_epi16(zbin, log_scale_vec);
267
0
  round = _mm_add_epi16(round, log_scale_vec);
268
0
  zbin = _mm_srli_epi16(zbin, log_scale);
269
0
  round = _mm_srli_epi16(round, log_scale);
270
0
  zbin = _mm_sub_epi16(zbin, one);
271
272
  // Do DC and first 15 AC.
273
0
  coeff0 = load_coefficients(coeff_ptr);
274
0
  coeff1 = load_coefficients(coeff_ptr + 8);
275
276
0
  coeff0_sign = _mm_srai_epi16(coeff0, 15);
277
0
  coeff1_sign = _mm_srai_epi16(coeff1, 15);
278
0
  qcoeff0 = invert_sign_sse2(coeff0, coeff0_sign);
279
0
  qcoeff1 = invert_sign_sse2(coeff1, coeff1_sign);
280
281
0
  update_mask0(&qcoeff0, &qcoeff1, threshold, iscan, &is_found0, &mask0);
282
283
0
  cmp_mask0 = _mm_cmpgt_epi16(qcoeff0, zbin);
284
0
  zbin = _mm_unpackhi_epi64(zbin, zbin);  // Switch DC to AC
285
0
  cmp_mask1 = _mm_cmpgt_epi16(qcoeff1, zbin);
286
287
0
  update_mask1(&cmp_mask0, &cmp_mask1, iscan, &is_found1, &mask1);
288
289
0
  threshold[0] = threshold[1];
290
0
  all_zero = _mm_or_si128(cmp_mask0, cmp_mask1);
291
0
  if (_mm_movemask_epi8(all_zero) == 0) {
292
0
    _mm_store_si128((__m128i *)(qcoeff_ptr), zero);
293
0
    _mm_store_si128((__m128i *)(qcoeff_ptr + 4), zero);
294
0
    _mm_store_si128((__m128i *)(qcoeff_ptr + 8), zero);
295
0
    _mm_store_si128((__m128i *)(qcoeff_ptr + 12), zero);
296
0
    _mm_store_si128((__m128i *)(dqcoeff_ptr), zero);
297
0
    _mm_store_si128((__m128i *)(dqcoeff_ptr + 4), zero);
298
0
    _mm_store_si128((__m128i *)(dqcoeff_ptr + 8), zero);
299
0
    _mm_store_si128((__m128i *)(dqcoeff_ptr + 12), zero);
300
0
    round = _mm_unpackhi_epi64(round, round);
301
0
    quant = _mm_unpackhi_epi64(quant, quant);
302
0
    shift = _mm_unpackhi_epi64(shift, shift);
303
0
    dequant = _mm_unpackhi_epi64(dequant, dequant);
304
0
  } else {
305
0
    calculate_qcoeff_log_scale(&qcoeff0, round, quant, &shift, &log_scale);
306
0
    round = _mm_unpackhi_epi64(round, round);
307
0
    quant = _mm_unpackhi_epi64(quant, quant);
308
0
    shift = _mm_unpackhi_epi64(shift, shift);
309
0
    calculate_qcoeff_log_scale(&qcoeff1, round, quant, &shift, &log_scale);
310
311
    // Reinsert signs
312
0
    qcoeff0 = invert_sign_sse2(qcoeff0, coeff0_sign);
313
0
    qcoeff1 = invert_sign_sse2(qcoeff1, coeff1_sign);
314
315
    // Mask out zbin threshold coeffs
316
0
    qcoeff0 = _mm_and_si128(qcoeff0, cmp_mask0);
317
0
    qcoeff1 = _mm_and_si128(qcoeff1, cmp_mask1);
318
319
0
    store_coefficients(qcoeff0, qcoeff_ptr);
320
0
    store_coefficients(qcoeff1, qcoeff_ptr + 8);
321
322
0
    calculate_dqcoeff_and_store_log_scale(qcoeff0, dequant, zero, dqcoeff_ptr,
323
0
                                          &log_scale);
324
0
    dequant = _mm_unpackhi_epi64(dequant, dequant);
325
0
    calculate_dqcoeff_and_store_log_scale(qcoeff1, dequant, zero,
326
0
                                          dqcoeff_ptr + 8, &log_scale);
327
0
  }
328
329
  // AC only loop.
330
0
  while (index < n_coeffs) {
331
0
    coeff0 = load_coefficients(coeff_ptr + index);
332
0
    coeff1 = load_coefficients(coeff_ptr + index + 8);
333
334
0
    coeff0_sign = _mm_srai_epi16(coeff0, 15);
335
0
    coeff1_sign = _mm_srai_epi16(coeff1, 15);
336
0
    qcoeff0 = invert_sign_sse2(coeff0, coeff0_sign);
337
0
    qcoeff1 = invert_sign_sse2(coeff1, coeff1_sign);
338
339
0
    update_mask0(&qcoeff0, &qcoeff1, threshold, iscan + index, &is_found0,
340
0
                 &mask0);
341
342
0
    cmp_mask0 = _mm_cmpgt_epi16(qcoeff0, zbin);
343
0
    cmp_mask1 = _mm_cmpgt_epi16(qcoeff1, zbin);
344
345
0
    update_mask1(&cmp_mask0, &cmp_mask1, iscan + index, &is_found1, &mask1);
346
347
0
    all_zero = _mm_or_si128(cmp_mask0, cmp_mask1);
348
0
    if (_mm_movemask_epi8(all_zero) == 0) {
349
0
      _mm_store_si128((__m128i *)(qcoeff_ptr + index), zero);
350
0
      _mm_store_si128((__m128i *)(qcoeff_ptr + index + 4), zero);
351
0
      _mm_store_si128((__m128i *)(qcoeff_ptr + index + 8), zero);
352
0
      _mm_store_si128((__m128i *)(qcoeff_ptr + index + 12), zero);
353
0
      _mm_store_si128((__m128i *)(dqcoeff_ptr + index), zero);
354
0
      _mm_store_si128((__m128i *)(dqcoeff_ptr + index + 4), zero);
355
0
      _mm_store_si128((__m128i *)(dqcoeff_ptr + index + 8), zero);
356
0
      _mm_store_si128((__m128i *)(dqcoeff_ptr + index + 12), zero);
357
0
      index += 16;
358
0
      continue;
359
0
    }
360
0
    calculate_qcoeff_log_scale(&qcoeff0, round, quant, &shift, &log_scale);
361
0
    calculate_qcoeff_log_scale(&qcoeff1, round, quant, &shift, &log_scale);
362
363
0
    qcoeff0 = invert_sign_sse2(qcoeff0, coeff0_sign);
364
0
    qcoeff1 = invert_sign_sse2(qcoeff1, coeff1_sign);
365
366
0
    qcoeff0 = _mm_and_si128(qcoeff0, cmp_mask0);
367
0
    qcoeff1 = _mm_and_si128(qcoeff1, cmp_mask1);
368
369
0
    store_coefficients(qcoeff0, qcoeff_ptr + index);
370
0
    store_coefficients(qcoeff1, qcoeff_ptr + index + 8);
371
372
0
    calculate_dqcoeff_and_store_log_scale(qcoeff0, dequant, zero,
373
0
                                          dqcoeff_ptr + index, &log_scale);
374
0
    calculate_dqcoeff_and_store_log_scale(qcoeff1, dequant, zero,
375
0
                                          dqcoeff_ptr + index + 8, &log_scale);
376
0
    index += 16;
377
0
  }
378
0
  if (is_found0) non_zero_count = calculate_non_zero_count(mask0);
379
0
  if (is_found1)
380
0
    non_zero_count_prescan_add_zero = calculate_non_zero_count(mask1);
381
382
0
  for (int i = non_zero_count_prescan_add_zero - 1; i >= non_zero_count; i--) {
383
0
    const int rc = scan[i];
384
0
    qcoeff_ptr[rc] = 0;
385
0
    dqcoeff_ptr[rc] = 0;
386
0
  }
387
388
0
  for (int i = non_zero_count - 1; i >= 0; i--) {
389
0
    const int rc = scan[i];
390
0
    if (qcoeff_ptr[rc]) {
391
0
      eob = i;
392
0
      break;
393
0
    }
394
0
  }
395
396
0
  *eob_ptr = eob + 1;
397
0
#if SKIP_EOB_FACTOR_ADJUST
398
  // TODO(Aniket): Experiment the following loop with intrinsic by combining
399
  // with the quantization loop above
400
0
  for (int i = 0; i < non_zero_count; i++) {
401
0
    const int rc = scan[i];
402
0
    const int qcoeff = qcoeff_ptr[rc];
403
0
    if (qcoeff) {
404
0
      first = i;
405
0
      break;
406
0
    }
407
0
  }
408
0
  if ((*eob_ptr - 1) >= 0 && first == (*eob_ptr - 1)) {
409
0
    const int rc = scan[(*eob_ptr - 1)];
410
0
    if (qcoeff_ptr[rc] == 1 || qcoeff_ptr[rc] == -1) {
411
0
      const int coeff = coeff_ptr[rc] * wt;
412
0
      const int coeff_sign = AOMSIGN(coeff);
413
0
      const int abs_coeff = (coeff ^ coeff_sign) - coeff_sign;
414
0
      const int factor = EOB_FACTOR + SKIP_EOB_FACTOR_ADJUST;
415
0
      const int prescan_add_val =
416
0
          ROUND_POWER_OF_TWO(dequant_ptr[rc != 0] * factor, 7);
417
0
      if (abs_coeff < (zbins[rc != 0] * (1 << AOM_QM_BITS) + prescan_add_val)) {
418
0
        qcoeff_ptr[rc] = 0;
419
0
        dqcoeff_ptr[rc] = 0;
420
0
        *eob_ptr = 0;
421
0
      }
422
0
    }
423
0
  }
424
0
#endif
425
0
}
426
427
void aom_quantize_b_64x64_adaptive_sse2(
428
    const tran_low_t *coeff_ptr, intptr_t n_coeffs, const int16_t *zbin_ptr,
429
    const int16_t *round_ptr, const int16_t *quant_ptr,
430
    const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr,
431
    tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr,
432
0
    const int16_t *scan, const int16_t *iscan) {
433
0
  int index = 16;
434
0
  const int log_scale = 2;
435
0
  int non_zero_count = 0;
436
0
  int non_zero_count_prescan_add_zero = 0;
437
0
  int is_found0 = 0, is_found1 = 0;
438
0
  int eob = -1;
439
0
  const __m128i zero = _mm_setzero_si128();
440
0
  const __m128i one = _mm_set1_epi16(1);
441
0
  const __m128i log_scale_vec = _mm_set1_epi16(log_scale);
442
0
  __m128i zbin, round, quant, dequant, shift;
443
0
  __m128i coeff0, coeff1, coeff0_sign, coeff1_sign;
444
0
  __m128i qcoeff0, qcoeff1;
445
0
  __m128i cmp_mask0, cmp_mask1;
446
0
  __m128i all_zero;
447
0
  __m128i mask0 = zero, mask1 = zero;
448
449
0
  const int zbins[2] = { ROUND_POWER_OF_TWO(zbin_ptr[0], log_scale),
450
0
                         ROUND_POWER_OF_TWO(zbin_ptr[1], log_scale) };
451
0
  int prescan_add[2];
452
0
  int thresh[4];
453
0
  const qm_val_t wt = (1 << AOM_QM_BITS);
454
0
  for (int i = 0; i < 2; ++i) {
455
0
    prescan_add[i] = ROUND_POWER_OF_TWO(dequant_ptr[i] * EOB_FACTOR, 7);
456
0
    thresh[i] = (zbins[i] * wt + prescan_add[i]) - 1;
457
0
  }
458
0
  thresh[2] = thresh[3] = thresh[1];
459
0
  __m128i threshold[2];
460
0
  threshold[0] = _mm_loadu_si128((__m128i *)&thresh[0]);
461
0
  threshold[1] = _mm_unpackhi_epi64(threshold[0], threshold[0]);
462
463
0
#if SKIP_EOB_FACTOR_ADJUST
464
0
  int first = -1;
465
0
#endif
466
  // Setup global values.
467
0
  zbin = _mm_load_si128((const __m128i *)zbin_ptr);
468
0
  round = _mm_load_si128((const __m128i *)round_ptr);
469
0
  quant = _mm_load_si128((const __m128i *)quant_ptr);
470
0
  dequant = _mm_load_si128((const __m128i *)dequant_ptr);
471
0
  shift = _mm_load_si128((const __m128i *)quant_shift_ptr);
472
473
  // Shift with rounding.
474
0
  zbin = _mm_add_epi16(zbin, log_scale_vec);
475
0
  round = _mm_add_epi16(round, log_scale_vec);
476
0
  zbin = _mm_srli_epi16(zbin, log_scale);
477
0
  round = _mm_srli_epi16(round, log_scale);
478
0
  zbin = _mm_sub_epi16(zbin, one);
479
480
  // Do DC and first 15 AC.
481
0
  coeff0 = load_coefficients(coeff_ptr);
482
0
  coeff1 = load_coefficients(coeff_ptr + 8);
483
484
0
  coeff0_sign = _mm_srai_epi16(coeff0, 15);
485
0
  coeff1_sign = _mm_srai_epi16(coeff1, 15);
486
0
  qcoeff0 = invert_sign_sse2(coeff0, coeff0_sign);
487
0
  qcoeff1 = invert_sign_sse2(coeff1, coeff1_sign);
488
489
0
  update_mask0(&qcoeff0, &qcoeff1, threshold, iscan, &is_found0, &mask0);
490
491
0
  cmp_mask0 = _mm_cmpgt_epi16(qcoeff0, zbin);
492
0
  zbin = _mm_unpackhi_epi64(zbin, zbin);  // Switch DC to AC
493
0
  cmp_mask1 = _mm_cmpgt_epi16(qcoeff1, zbin);
494
495
0
  update_mask1(&cmp_mask0, &cmp_mask1, iscan, &is_found1, &mask1);
496
497
0
  threshold[0] = threshold[1];
498
0
  all_zero = _mm_or_si128(cmp_mask0, cmp_mask1);
499
0
  if (_mm_movemask_epi8(all_zero) == 0) {
500
0
    _mm_store_si128((__m128i *)(qcoeff_ptr), zero);
501
0
    _mm_store_si128((__m128i *)(qcoeff_ptr + 4), zero);
502
0
    _mm_store_si128((__m128i *)(qcoeff_ptr + 8), zero);
503
0
    _mm_store_si128((__m128i *)(qcoeff_ptr + 12), zero);
504
0
    _mm_store_si128((__m128i *)(dqcoeff_ptr), zero);
505
0
    _mm_store_si128((__m128i *)(dqcoeff_ptr + 4), zero);
506
0
    _mm_store_si128((__m128i *)(dqcoeff_ptr + 8), zero);
507
0
    _mm_store_si128((__m128i *)(dqcoeff_ptr + 12), zero);
508
0
    round = _mm_unpackhi_epi64(round, round);
509
0
    quant = _mm_unpackhi_epi64(quant, quant);
510
0
    shift = _mm_unpackhi_epi64(shift, shift);
511
0
    dequant = _mm_unpackhi_epi64(dequant, dequant);
512
0
  } else {
513
0
    calculate_qcoeff_log_scale(&qcoeff0, round, quant, &shift, &log_scale);
514
0
    round = _mm_unpackhi_epi64(round, round);
515
0
    quant = _mm_unpackhi_epi64(quant, quant);
516
0
    shift = _mm_unpackhi_epi64(shift, shift);
517
0
    calculate_qcoeff_log_scale(&qcoeff1, round, quant, &shift, &log_scale);
518
519
    // Reinsert signs
520
0
    qcoeff0 = invert_sign_sse2(qcoeff0, coeff0_sign);
521
0
    qcoeff1 = invert_sign_sse2(qcoeff1, coeff1_sign);
522
523
    // Mask out zbin threshold coeffs
524
0
    qcoeff0 = _mm_and_si128(qcoeff0, cmp_mask0);
525
0
    qcoeff1 = _mm_and_si128(qcoeff1, cmp_mask1);
526
527
0
    store_coefficients(qcoeff0, qcoeff_ptr);
528
0
    store_coefficients(qcoeff1, qcoeff_ptr + 8);
529
530
0
    calculate_dqcoeff_and_store_log_scale(qcoeff0, dequant, zero, dqcoeff_ptr,
531
0
                                          &log_scale);
532
0
    dequant = _mm_unpackhi_epi64(dequant, dequant);
533
0
    calculate_dqcoeff_and_store_log_scale(qcoeff1, dequant, zero,
534
0
                                          dqcoeff_ptr + 8, &log_scale);
535
0
  }
536
537
  // AC only loop.
538
0
  while (index < n_coeffs) {
539
0
    coeff0 = load_coefficients(coeff_ptr + index);
540
0
    coeff1 = load_coefficients(coeff_ptr + index + 8);
541
542
0
    coeff0_sign = _mm_srai_epi16(coeff0, 15);
543
0
    coeff1_sign = _mm_srai_epi16(coeff1, 15);
544
0
    qcoeff0 = invert_sign_sse2(coeff0, coeff0_sign);
545
0
    qcoeff1 = invert_sign_sse2(coeff1, coeff1_sign);
546
547
0
    update_mask0(&qcoeff0, &qcoeff1, threshold, iscan + index, &is_found0,
548
0
                 &mask0);
549
550
0
    cmp_mask0 = _mm_cmpgt_epi16(qcoeff0, zbin);
551
0
    cmp_mask1 = _mm_cmpgt_epi16(qcoeff1, zbin);
552
553
0
    update_mask1(&cmp_mask0, &cmp_mask1, iscan + index, &is_found1, &mask1);
554
555
0
    all_zero = _mm_or_si128(cmp_mask0, cmp_mask1);
556
0
    if (_mm_movemask_epi8(all_zero) == 0) {
557
0
      _mm_store_si128((__m128i *)(qcoeff_ptr + index), zero);
558
0
      _mm_store_si128((__m128i *)(qcoeff_ptr + index + 4), zero);
559
0
      _mm_store_si128((__m128i *)(qcoeff_ptr + index + 8), zero);
560
0
      _mm_store_si128((__m128i *)(qcoeff_ptr + index + 12), zero);
561
0
      _mm_store_si128((__m128i *)(dqcoeff_ptr + index), zero);
562
0
      _mm_store_si128((__m128i *)(dqcoeff_ptr + index + 4), zero);
563
0
      _mm_store_si128((__m128i *)(dqcoeff_ptr + index + 8), zero);
564
0
      _mm_store_si128((__m128i *)(dqcoeff_ptr + index + 12), zero);
565
0
      index += 16;
566
0
      continue;
567
0
    }
568
0
    calculate_qcoeff_log_scale(&qcoeff0, round, quant, &shift, &log_scale);
569
0
    calculate_qcoeff_log_scale(&qcoeff1, round, quant, &shift, &log_scale);
570
571
0
    qcoeff0 = invert_sign_sse2(qcoeff0, coeff0_sign);
572
0
    qcoeff1 = invert_sign_sse2(qcoeff1, coeff1_sign);
573
574
0
    qcoeff0 = _mm_and_si128(qcoeff0, cmp_mask0);
575
0
    qcoeff1 = _mm_and_si128(qcoeff1, cmp_mask1);
576
577
0
    store_coefficients(qcoeff0, qcoeff_ptr + index);
578
0
    store_coefficients(qcoeff1, qcoeff_ptr + index + 8);
579
580
0
    calculate_dqcoeff_and_store_log_scale(qcoeff0, dequant, zero,
581
0
                                          dqcoeff_ptr + index, &log_scale);
582
0
    calculate_dqcoeff_and_store_log_scale(qcoeff1, dequant, zero,
583
0
                                          dqcoeff_ptr + index + 8, &log_scale);
584
0
    index += 16;
585
0
  }
586
0
  if (is_found0) non_zero_count = calculate_non_zero_count(mask0);
587
0
  if (is_found1)
588
0
    non_zero_count_prescan_add_zero = calculate_non_zero_count(mask1);
589
590
0
  for (int i = non_zero_count_prescan_add_zero - 1; i >= non_zero_count; i--) {
591
0
    const int rc = scan[i];
592
0
    qcoeff_ptr[rc] = 0;
593
0
    dqcoeff_ptr[rc] = 0;
594
0
  }
595
596
0
  for (int i = non_zero_count - 1; i >= 0; i--) {
597
0
    const int rc = scan[i];
598
0
    if (qcoeff_ptr[rc]) {
599
0
      eob = i;
600
0
      break;
601
0
    }
602
0
  }
603
604
0
  *eob_ptr = eob + 1;
605
0
#if SKIP_EOB_FACTOR_ADJUST
606
  // TODO(Aniket): Experiment the following loop with intrinsic by combining
607
  // with the quantization loop above
608
0
  for (int i = 0; i < non_zero_count; i++) {
609
0
    const int rc = scan[i];
610
0
    const int qcoeff = qcoeff_ptr[rc];
611
0
    if (qcoeff) {
612
0
      first = i;
613
0
      break;
614
0
    }
615
0
  }
616
0
  if ((*eob_ptr - 1) >= 0 && first == (*eob_ptr - 1)) {
617
0
    const int rc = scan[(*eob_ptr - 1)];
618
0
    if (qcoeff_ptr[rc] == 1 || qcoeff_ptr[rc] == -1) {
619
0
      const int coeff = coeff_ptr[rc] * wt;
620
0
      const int coeff_sign = AOMSIGN(coeff);
621
0
      const int abs_coeff = (coeff ^ coeff_sign) - coeff_sign;
622
0
      const int factor = EOB_FACTOR + SKIP_EOB_FACTOR_ADJUST;
623
0
      const int prescan_add_val =
624
0
          ROUND_POWER_OF_TWO(dequant_ptr[rc != 0] * factor, 7);
625
0
      if (abs_coeff < (zbins[rc != 0] * (1 << AOM_QM_BITS) + prescan_add_val)) {
626
0
        qcoeff_ptr[rc] = 0;
627
0
        dqcoeff_ptr[rc] = 0;
628
0
        *eob_ptr = 0;
629
0
      }
630
0
    }
631
0
  }
632
0
#endif
633
0
}