Coverage Report

Created: 2026-09-07 06:44

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/aom/av1/encoder/x86/av1_quantize_avx2.c
Line
Count
Source
1
/*
2
 * Copyright (c) 2017, Alliance for Open Media. All rights reserved.
3
 *
4
 * This source code is subject to the terms of the BSD 2 Clause License and
5
 * the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
6
 * was not distributed with this source code in the LICENSE file, you can
7
 * obtain it at www.aomedia.org/license/software. If the Alliance for Open
8
 * Media Patent License 1.0 was not distributed with this source code in the
9
 * PATENTS file, you can obtain it at www.aomedia.org/license/patent.
10
 */
11
12
#include <immintrin.h>
13
14
#include "config/av1_rtcd.h"
15
16
#include "aom/aom_integer.h"
17
#include "aom_dsp/aom_dsp_common.h"
18
19
0
static inline void write_zero(tran_low_t *qcoeff) {
20
0
  const __m256i zero = _mm256_setzero_si256();
21
0
  _mm256_storeu_si256((__m256i *)qcoeff, zero);
22
0
  _mm256_storeu_si256((__m256i *)qcoeff + 1, zero);
23
0
}
24
25
0
static inline void init_one_qp(const __m128i *p, __m256i *qp) {
26
0
  const __m128i ac = _mm_unpackhi_epi64(*p, *p);
27
0
  *qp = _mm256_insertf128_si256(_mm256_castsi128_si256(*p), ac, 1);
28
0
}
29
30
static inline void init_qp(const int16_t *round_ptr, const int16_t *quant_ptr,
31
                           const int16_t *dequant_ptr, int log_scale,
32
0
                           __m256i *thr, __m256i *qp) {
33
0
  __m128i round = _mm_loadu_si128((const __m128i *)round_ptr);
34
0
  const __m128i quant = _mm_loadu_si128((const __m128i *)quant_ptr);
35
0
  const __m128i dequant = _mm_loadu_si128((const __m128i *)dequant_ptr);
36
37
0
  if (log_scale > 0) {
38
0
    const __m128i rnd = _mm_set1_epi16((int16_t)1 << (log_scale - 1));
39
0
    round = _mm_add_epi16(round, rnd);
40
0
    round = _mm_srai_epi16(round, log_scale);
41
0
  }
42
43
0
  init_one_qp(&round, &qp[0]);
44
0
  init_one_qp(&quant, &qp[1]);
45
46
0
  if (log_scale == 1) {
47
0
    qp[1] = _mm256_slli_epi16(qp[1], log_scale);
48
0
  }
49
50
0
  init_one_qp(&dequant, &qp[2]);
51
0
  *thr = _mm256_srai_epi16(qp[2], 1 + log_scale);
52
  // Subtracting 1 here eliminates a _mm256_cmpeq_epi16() instruction when
53
  // calculating the zbin mask.
54
0
  *thr = _mm256_sub_epi16(*thr, _mm256_set1_epi16(1));
55
0
}
56
57
0
static inline void update_qp(__m256i *thr, __m256i *qp) {
58
0
  qp[0] = _mm256_permute2x128_si256(qp[0], qp[0], 0x11);
59
0
  qp[1] = _mm256_permute2x128_si256(qp[1], qp[1], 0x11);
60
0
  qp[2] = _mm256_permute2x128_si256(qp[2], qp[2], 0x11);
61
0
  *thr = _mm256_permute2x128_si256(*thr, *thr, 0x11);
62
0
}
63
64
0
static inline __m256i load_coefficients_avx2(const tran_low_t *coeff_ptr) {
65
0
  const __m256i coeff1 = _mm256_load_si256((__m256i *)coeff_ptr);
66
0
  const __m256i coeff2 = _mm256_load_si256((__m256i *)(coeff_ptr + 8));
67
0
  return _mm256_packs_epi32(coeff1, coeff2);
68
0
}
69
70
static inline void store_coefficients_avx2(__m256i coeff_vals,
71
0
                                           tran_low_t *coeff_ptr) {
72
0
  __m256i coeff_sign = _mm256_srai_epi16(coeff_vals, 15);
73
0
  __m256i coeff_vals_lo = _mm256_unpacklo_epi16(coeff_vals, coeff_sign);
74
0
  __m256i coeff_vals_hi = _mm256_unpackhi_epi16(coeff_vals, coeff_sign);
75
0
  _mm256_store_si256((__m256i *)coeff_ptr, coeff_vals_lo);
76
0
  _mm256_store_si256((__m256i *)(coeff_ptr + 8), coeff_vals_hi);
77
0
}
78
79
0
static inline uint16_t quant_gather_eob(__m256i eob) {
80
0
  const __m128i eob_lo = _mm256_castsi256_si128(eob);
81
0
  const __m128i eob_hi = _mm256_extractf128_si256(eob, 1);
82
0
  __m128i eob_s = _mm_max_epi16(eob_lo, eob_hi);
83
0
  eob_s = _mm_subs_epu16(_mm_set1_epi16(INT16_MAX), eob_s);
84
0
  eob_s = _mm_minpos_epu16(eob_s);
85
0
  return INT16_MAX - _mm_extract_epi16(eob_s, 0);
86
0
}
87
88
0
static inline int16_t accumulate_eob256(__m256i eob256) {
89
0
  const __m128i eob_lo = _mm256_castsi256_si128(eob256);
90
0
  const __m128i eob_hi = _mm256_extractf128_si256(eob256, 1);
91
0
  __m128i eob = _mm_max_epi16(eob_lo, eob_hi);
92
0
  __m128i eob_shuffled = _mm_shuffle_epi32(eob, 0xe);
93
0
  eob = _mm_max_epi16(eob, eob_shuffled);
94
0
  eob_shuffled = _mm_shufflelo_epi16(eob, 0xe);
95
0
  eob = _mm_max_epi16(eob, eob_shuffled);
96
0
  eob_shuffled = _mm_shufflelo_epi16(eob, 0x1);
97
0
  eob = _mm_max_epi16(eob, eob_shuffled);
98
0
  return _mm_extract_epi16(eob, 1);
99
0
}
100
101
static AOM_FORCE_INLINE void quantize_lp_16_first(
102
    const int16_t *coeff_ptr, const int16_t *iscan_ptr, int16_t *qcoeff_ptr,
103
    int16_t *dqcoeff_ptr, __m256i *round256, __m256i *quant256,
104
0
    __m256i *dequant256, __m256i *eob) {
105
0
  const __m256i coeff = _mm256_loadu_si256((const __m256i *)coeff_ptr);
106
0
  const __m256i abs_coeff = _mm256_abs_epi16(coeff);
107
0
  const __m256i tmp_rnd = _mm256_adds_epi16(abs_coeff, *round256);
108
0
  const __m256i abs_qcoeff = _mm256_mulhi_epi16(tmp_rnd, *quant256);
109
0
  const __m256i qcoeff = _mm256_sign_epi16(abs_qcoeff, coeff);
110
0
  const __m256i dqcoeff = _mm256_mullo_epi16(qcoeff, *dequant256);
111
0
  const __m256i nz_mask =
112
0
      _mm256_cmpgt_epi16(abs_qcoeff, _mm256_setzero_si256());
113
114
0
  _mm256_storeu_si256((__m256i *)qcoeff_ptr, qcoeff);
115
0
  _mm256_storeu_si256((__m256i *)dqcoeff_ptr, dqcoeff);
116
117
0
  const __m256i iscan = _mm256_loadu_si256((const __m256i *)iscan_ptr);
118
0
  const __m256i iscan_plus1 = _mm256_sub_epi16(iscan, nz_mask);
119
0
  const __m256i nz_iscan = _mm256_and_si256(iscan_plus1, nz_mask);
120
0
  *eob = _mm256_max_epi16(*eob, nz_iscan);
121
0
}
122
123
static AOM_FORCE_INLINE void quantize_lp_16(
124
    const int16_t *coeff_ptr, intptr_t n_coeffs, const int16_t *iscan_ptr,
125
    int16_t *qcoeff_ptr, int16_t *dqcoeff_ptr, __m256i *round256,
126
0
    __m256i *quant256, __m256i *dequant256, __m256i *eob) {
127
0
  const __m256i coeff =
128
0
      _mm256_loadu_si256((const __m256i *)(coeff_ptr + n_coeffs));
129
0
  const __m256i abs_coeff = _mm256_abs_epi16(coeff);
130
0
  const __m256i tmp_rnd = _mm256_adds_epi16(abs_coeff, *round256);
131
0
  const __m256i abs_qcoeff = _mm256_mulhi_epi16(tmp_rnd, *quant256);
132
0
  const __m256i qcoeff = _mm256_sign_epi16(abs_qcoeff, coeff);
133
0
  const __m256i dqcoeff = _mm256_mullo_epi16(qcoeff, *dequant256);
134
0
  const __m256i nz_mask =
135
0
      _mm256_cmpgt_epi16(abs_qcoeff, _mm256_setzero_si256());
136
137
0
  _mm256_storeu_si256((__m256i *)(qcoeff_ptr + n_coeffs), qcoeff);
138
0
  _mm256_storeu_si256((__m256i *)(dqcoeff_ptr + n_coeffs), dqcoeff);
139
140
0
  const __m256i iscan =
141
0
      _mm256_loadu_si256((const __m256i *)(iscan_ptr + n_coeffs));
142
0
  const __m256i iscan_plus1 = _mm256_sub_epi16(iscan, nz_mask);
143
0
  const __m256i nz_iscan = _mm256_and_si256(iscan_plus1, nz_mask);
144
0
  *eob = _mm256_max_epi16(*eob, nz_iscan);
145
0
}
146
147
void av1_quantize_lp_avx2(const int16_t *coeff_ptr, intptr_t n_coeffs,
148
                          const int16_t *round_ptr, const int16_t *quant_ptr,
149
                          int16_t *qcoeff_ptr, int16_t *dqcoeff_ptr,
150
                          const int16_t *dequant_ptr, uint16_t *eob_ptr,
151
0
                          const int16_t *scan, const int16_t *iscan) {
152
0
  (void)scan;
153
0
  __m256i eob256 = _mm256_setzero_si256();
154
155
  // Setup global values.
156
0
  __m256i round256 =
157
0
      _mm256_castsi128_si256(_mm_load_si128((const __m128i *)round_ptr));
158
0
  __m256i quant256 =
159
0
      _mm256_castsi128_si256(_mm_load_si128((const __m128i *)quant_ptr));
160
0
  __m256i dequant256 =
161
0
      _mm256_castsi128_si256(_mm_load_si128((const __m128i *)dequant_ptr));
162
163
  // Populate upper AC values.
164
0
  round256 = _mm256_permute4x64_epi64(round256, 0x54);
165
0
  quant256 = _mm256_permute4x64_epi64(quant256, 0x54);
166
0
  dequant256 = _mm256_permute4x64_epi64(dequant256, 0x54);
167
168
  // Process DC and the first 15 AC coeffs.
169
0
  quantize_lp_16_first(coeff_ptr, iscan, qcoeff_ptr, dqcoeff_ptr, &round256,
170
0
                       &quant256, &dequant256, &eob256);
171
172
0
  if (n_coeffs > 16) {
173
    // Overwrite the DC constants with AC constants
174
0
    dequant256 = _mm256_permute2x128_si256(dequant256, dequant256, 0x31);
175
0
    quant256 = _mm256_permute2x128_si256(quant256, quant256, 0x31);
176
0
    round256 = _mm256_permute2x128_si256(round256, round256, 0x31);
177
178
    // AC only loop.
179
0
    for (int idx = 16; idx < n_coeffs; idx += 16) {
180
0
      quantize_lp_16(coeff_ptr, idx, iscan, qcoeff_ptr, dqcoeff_ptr, &round256,
181
0
                     &quant256, &dequant256, &eob256);
182
0
    }
183
0
  }
184
185
0
  *eob_ptr = accumulate_eob256(eob256);
186
0
}
187
188
static AOM_FORCE_INLINE __m256i get_max_lane_eob(const int16_t *iscan,
189
                                                 __m256i v_eobmax,
190
0
                                                 __m256i v_mask) {
191
0
  const __m256i v_iscan = _mm256_loadu_si256((const __m256i *)iscan);
192
0
  const __m256i v_iscan_perm = _mm256_permute4x64_epi64(v_iscan, 0xD8);
193
0
  const __m256i v_iscan_plus1 = _mm256_sub_epi16(v_iscan_perm, v_mask);
194
0
  const __m256i v_nz_iscan = _mm256_and_si256(v_iscan_plus1, v_mask);
195
0
  return _mm256_max_epi16(v_eobmax, v_nz_iscan);
196
0
}
197
198
static AOM_FORCE_INLINE void quantize_fp_16(
199
    const __m256i *thr, const __m256i *qp, const tran_low_t *coeff_ptr,
200
    const int16_t *iscan_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr,
201
0
    __m256i *eob) {
202
0
  const __m256i coeff = load_coefficients_avx2(coeff_ptr);
203
0
  const __m256i abs_coeff = _mm256_abs_epi16(coeff);
204
0
  const __m256i mask = _mm256_cmpgt_epi16(abs_coeff, *thr);
205
0
  const int nzflag = _mm256_movemask_epi8(mask);
206
207
0
  if (nzflag) {
208
0
    const __m256i tmp_rnd = _mm256_adds_epi16(abs_coeff, qp[0]);
209
0
    const __m256i abs_q = _mm256_mulhi_epi16(tmp_rnd, qp[1]);
210
0
    const __m256i q = _mm256_sign_epi16(abs_q, coeff);
211
0
    const __m256i dq = _mm256_mullo_epi16(q, qp[2]);
212
0
    const __m256i nz_mask = _mm256_cmpgt_epi16(abs_q, _mm256_setzero_si256());
213
214
0
    store_coefficients_avx2(q, qcoeff_ptr);
215
0
    store_coefficients_avx2(dq, dqcoeff_ptr);
216
217
0
    *eob = get_max_lane_eob(iscan_ptr, *eob, nz_mask);
218
0
  } else {
219
0
    write_zero(qcoeff_ptr);
220
0
    write_zero(dqcoeff_ptr);
221
0
  }
222
0
}
223
224
void av1_quantize_fp_avx2(const tran_low_t *coeff_ptr, intptr_t n_coeffs,
225
                          const int16_t *zbin_ptr, const int16_t *round_ptr,
226
                          const int16_t *quant_ptr,
227
                          const int16_t *quant_shift_ptr,
228
                          tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr,
229
                          const int16_t *dequant_ptr, uint16_t *eob_ptr,
230
0
                          const int16_t *scan_ptr, const int16_t *iscan_ptr) {
231
0
  (void)scan_ptr;
232
0
  (void)zbin_ptr;
233
0
  (void)quant_shift_ptr;
234
235
0
  const int log_scale = 0;
236
0
  const int step = 16;
237
0
  __m256i qp[3], thr;
238
0
  __m256i eob = _mm256_setzero_si256();
239
240
0
  init_qp(round_ptr, quant_ptr, dequant_ptr, log_scale, &thr, qp);
241
242
0
  quantize_fp_16(&thr, qp, coeff_ptr, iscan_ptr, qcoeff_ptr, dqcoeff_ptr, &eob);
243
244
0
  coeff_ptr += step;
245
0
  qcoeff_ptr += step;
246
0
  dqcoeff_ptr += step;
247
0
  iscan_ptr += step;
248
0
  n_coeffs -= step;
249
250
0
  update_qp(&thr, qp);
251
252
0
  while (n_coeffs > 0) {
253
0
    quantize_fp_16(&thr, qp, coeff_ptr, iscan_ptr, qcoeff_ptr, dqcoeff_ptr,
254
0
                   &eob);
255
256
0
    coeff_ptr += step;
257
0
    qcoeff_ptr += step;
258
0
    dqcoeff_ptr += step;
259
0
    iscan_ptr += step;
260
0
    n_coeffs -= step;
261
0
  }
262
0
  *eob_ptr = quant_gather_eob(eob);
263
0
}
264
265
static AOM_FORCE_INLINE void quantize_fp_32x32(
266
    const __m256i *thr, const __m256i *qp, const tran_low_t *coeff_ptr,
267
    const int16_t *iscan_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr,
268
0
    __m256i *eob) {
269
0
  const __m256i coeff = load_coefficients_avx2(coeff_ptr);
270
0
  const __m256i abs_coeff = _mm256_abs_epi16(coeff);
271
0
  const __m256i mask = _mm256_cmpgt_epi16(abs_coeff, *thr);
272
0
  const int nzflag = _mm256_movemask_epi8(mask);
273
274
0
  if (nzflag) {
275
0
    const __m256i tmp_rnd = _mm256_adds_epi16(abs_coeff, qp[0]);
276
0
    const __m256i abs_q = _mm256_mulhi_epu16(tmp_rnd, qp[1]);
277
0
    const __m256i q = _mm256_sign_epi16(abs_q, coeff);
278
0
    const __m256i abs_dq =
279
0
        _mm256_srli_epi16(_mm256_mullo_epi16(abs_q, qp[2]), 1);
280
0
    const __m256i nz_mask = _mm256_cmpgt_epi16(abs_q, _mm256_setzero_si256());
281
0
    const __m256i dq = _mm256_sign_epi16(abs_dq, coeff);
282
283
0
    store_coefficients_avx2(q, qcoeff_ptr);
284
0
    store_coefficients_avx2(dq, dqcoeff_ptr);
285
286
0
    *eob = get_max_lane_eob(iscan_ptr, *eob, nz_mask);
287
0
  } else {
288
0
    write_zero(qcoeff_ptr);
289
0
    write_zero(dqcoeff_ptr);
290
0
  }
291
0
}
292
293
void av1_quantize_fp_32x32_avx2(
294
    const tran_low_t *coeff_ptr, intptr_t n_coeffs, const int16_t *zbin_ptr,
295
    const int16_t *round_ptr, const int16_t *quant_ptr,
296
    const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr,
297
    tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr,
298
0
    const int16_t *scan_ptr, const int16_t *iscan_ptr) {
299
0
  (void)scan_ptr;
300
0
  (void)zbin_ptr;
301
0
  (void)quant_shift_ptr;
302
303
0
  const int log_scale = 1;
304
0
  const unsigned int step = 16;
305
0
  __m256i qp[3], thr;
306
0
  __m256i eob = _mm256_setzero_si256();
307
308
0
  init_qp(round_ptr, quant_ptr, dequant_ptr, log_scale, &thr, qp);
309
310
0
  quantize_fp_32x32(&thr, qp, coeff_ptr, iscan_ptr, qcoeff_ptr, dqcoeff_ptr,
311
0
                    &eob);
312
313
0
  coeff_ptr += step;
314
0
  qcoeff_ptr += step;
315
0
  dqcoeff_ptr += step;
316
0
  iscan_ptr += step;
317
0
  n_coeffs -= step;
318
319
0
  update_qp(&thr, qp);
320
321
0
  while (n_coeffs > 0) {
322
0
    quantize_fp_32x32(&thr, qp, coeff_ptr, iscan_ptr, qcoeff_ptr, dqcoeff_ptr,
323
0
                      &eob);
324
325
0
    coeff_ptr += step;
326
0
    qcoeff_ptr += step;
327
0
    dqcoeff_ptr += step;
328
0
    iscan_ptr += step;
329
0
    n_coeffs -= step;
330
0
  }
331
0
  *eob_ptr = quant_gather_eob(eob);
332
0
}
333
334
static inline void quantize_fp_64x64(const __m256i *thr, const __m256i *qp,
335
                                     const tran_low_t *coeff_ptr,
336
                                     const int16_t *iscan_ptr,
337
                                     tran_low_t *qcoeff_ptr,
338
0
                                     tran_low_t *dqcoeff_ptr, __m256i *eob) {
339
0
  const __m256i coeff = load_coefficients_avx2(coeff_ptr);
340
0
  const __m256i abs_coeff = _mm256_abs_epi16(coeff);
341
0
  const __m256i mask = _mm256_cmpgt_epi16(abs_coeff, *thr);
342
0
  const int nzflag = _mm256_movemask_epi8(mask);
343
344
0
  if (nzflag) {
345
0
    const __m256i tmp_rnd =
346
0
        _mm256_and_si256(_mm256_adds_epi16(abs_coeff, qp[0]), mask);
347
0
    const __m256i qh = _mm256_slli_epi16(_mm256_mulhi_epi16(tmp_rnd, qp[1]), 2);
348
0
    const __m256i ql =
349
0
        _mm256_srli_epi16(_mm256_mullo_epi16(tmp_rnd, qp[1]), 14);
350
0
    const __m256i abs_q = _mm256_or_si256(qh, ql);
351
0
    const __m256i dqh = _mm256_slli_epi16(_mm256_mulhi_epi16(abs_q, qp[2]), 14);
352
0
    const __m256i dql = _mm256_srli_epi16(_mm256_mullo_epi16(abs_q, qp[2]), 2);
353
0
    const __m256i abs_dq = _mm256_or_si256(dqh, dql);
354
0
    const __m256i q = _mm256_sign_epi16(abs_q, coeff);
355
0
    const __m256i dq = _mm256_sign_epi16(abs_dq, coeff);
356
    // Check the signed q/dq value here instead of the absolute value. When
357
    // dequant equals 4, the dequant threshold (*thr) becomes 0 after being
358
    // scaled down by (1 + log_scale). See init_qp(). When *thr is 0 and the
359
    // abs_coeff is 0, the nzflag will be set. As a result, the eob will be
360
    // incorrectly calculated. The psign instruction corrects the error by
361
    // zeroing out q/dq if coeff is zero.
362
0
    const __m256i z_mask = _mm256_cmpeq_epi16(dq, _mm256_setzero_si256());
363
0
    const __m256i nz_mask = _mm256_cmpeq_epi16(z_mask, _mm256_setzero_si256());
364
365
0
    store_coefficients_avx2(q, qcoeff_ptr);
366
0
    store_coefficients_avx2(dq, dqcoeff_ptr);
367
368
0
    *eob = get_max_lane_eob(iscan_ptr, *eob, nz_mask);
369
0
  } else {
370
0
    write_zero(qcoeff_ptr);
371
0
    write_zero(dqcoeff_ptr);
372
0
  }
373
0
}
374
375
void av1_quantize_fp_64x64_avx2(
376
    const tran_low_t *coeff_ptr, intptr_t n_coeffs, const int16_t *zbin_ptr,
377
    const int16_t *round_ptr, const int16_t *quant_ptr,
378
    const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr,
379
    tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr,
380
0
    const int16_t *scan_ptr, const int16_t *iscan_ptr) {
381
0
  (void)scan_ptr;
382
0
  (void)zbin_ptr;
383
0
  (void)quant_shift_ptr;
384
385
0
  const int log_scale = 2;
386
0
  const unsigned int step = 16;
387
0
  __m256i qp[3], thr;
388
0
  __m256i eob = _mm256_setzero_si256();
389
390
0
  init_qp(round_ptr, quant_ptr, dequant_ptr, log_scale, &thr, qp);
391
392
0
  quantize_fp_64x64(&thr, qp, coeff_ptr, iscan_ptr, qcoeff_ptr, dqcoeff_ptr,
393
0
                    &eob);
394
395
0
  coeff_ptr += step;
396
0
  qcoeff_ptr += step;
397
0
  dqcoeff_ptr += step;
398
0
  iscan_ptr += step;
399
0
  n_coeffs -= step;
400
401
0
  update_qp(&thr, qp);
402
403
0
  while (n_coeffs > 0) {
404
0
    quantize_fp_64x64(&thr, qp, coeff_ptr, iscan_ptr, qcoeff_ptr, dqcoeff_ptr,
405
0
                      &eob);
406
407
0
    coeff_ptr += step;
408
0
    qcoeff_ptr += step;
409
0
    dqcoeff_ptr += step;
410
0
    iscan_ptr += step;
411
0
    n_coeffs -= step;
412
0
  }
413
0
  *eob_ptr = quant_gather_eob(eob);
414
0
}