/src/aom/av1/encoder/x86/av1_quantize_avx2.c
Line | Count | Source |
1 | | /* |
2 | | * Copyright (c) 2017, Alliance for Open Media. All rights reserved. |
3 | | * |
4 | | * This source code is subject to the terms of the BSD 2 Clause License and |
5 | | * the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License |
6 | | * was not distributed with this source code in the LICENSE file, you can |
7 | | * obtain it at www.aomedia.org/license/software. If the Alliance for Open |
8 | | * Media Patent License 1.0 was not distributed with this source code in the |
9 | | * PATENTS file, you can obtain it at www.aomedia.org/license/patent. |
10 | | */ |
11 | | |
12 | | #include <immintrin.h> |
13 | | |
14 | | #include "config/av1_rtcd.h" |
15 | | |
16 | | #include "aom/aom_integer.h" |
17 | | #include "aom_dsp/aom_dsp_common.h" |
18 | | |
19 | 0 | static inline void write_zero(tran_low_t *qcoeff) { |
20 | 0 | const __m256i zero = _mm256_setzero_si256(); |
21 | 0 | _mm256_storeu_si256((__m256i *)qcoeff, zero); |
22 | 0 | _mm256_storeu_si256((__m256i *)qcoeff + 1, zero); |
23 | 0 | } |
24 | | |
25 | 0 | static inline void init_one_qp(const __m128i *p, __m256i *qp) { |
26 | 0 | const __m128i ac = _mm_unpackhi_epi64(*p, *p); |
27 | 0 | *qp = _mm256_insertf128_si256(_mm256_castsi128_si256(*p), ac, 1); |
28 | 0 | } |
29 | | |
30 | | static inline void init_qp(const int16_t *round_ptr, const int16_t *quant_ptr, |
31 | | const int16_t *dequant_ptr, int log_scale, |
32 | 0 | __m256i *thr, __m256i *qp) { |
33 | 0 | __m128i round = _mm_loadu_si128((const __m128i *)round_ptr); |
34 | 0 | const __m128i quant = _mm_loadu_si128((const __m128i *)quant_ptr); |
35 | 0 | const __m128i dequant = _mm_loadu_si128((const __m128i *)dequant_ptr); |
36 | |
|
37 | 0 | if (log_scale > 0) { |
38 | 0 | const __m128i rnd = _mm_set1_epi16((int16_t)1 << (log_scale - 1)); |
39 | 0 | round = _mm_add_epi16(round, rnd); |
40 | 0 | round = _mm_srai_epi16(round, log_scale); |
41 | 0 | } |
42 | |
|
43 | 0 | init_one_qp(&round, &qp[0]); |
44 | 0 | init_one_qp(&quant, &qp[1]); |
45 | |
|
46 | 0 | if (log_scale == 1) { |
47 | 0 | qp[1] = _mm256_slli_epi16(qp[1], log_scale); |
48 | 0 | } |
49 | |
|
50 | 0 | init_one_qp(&dequant, &qp[2]); |
51 | 0 | *thr = _mm256_srai_epi16(qp[2], 1 + log_scale); |
52 | | // Subtracting 1 here eliminates a _mm256_cmpeq_epi16() instruction when |
53 | | // calculating the zbin mask. |
54 | 0 | *thr = _mm256_sub_epi16(*thr, _mm256_set1_epi16(1)); |
55 | 0 | } |
56 | | |
57 | 0 | static inline void update_qp(__m256i *thr, __m256i *qp) { |
58 | 0 | qp[0] = _mm256_permute2x128_si256(qp[0], qp[0], 0x11); |
59 | 0 | qp[1] = _mm256_permute2x128_si256(qp[1], qp[1], 0x11); |
60 | 0 | qp[2] = _mm256_permute2x128_si256(qp[2], qp[2], 0x11); |
61 | 0 | *thr = _mm256_permute2x128_si256(*thr, *thr, 0x11); |
62 | 0 | } |
63 | | |
64 | 0 | static inline __m256i load_coefficients_avx2(const tran_low_t *coeff_ptr) { |
65 | 0 | const __m256i coeff1 = _mm256_load_si256((__m256i *)coeff_ptr); |
66 | 0 | const __m256i coeff2 = _mm256_load_si256((__m256i *)(coeff_ptr + 8)); |
67 | 0 | return _mm256_packs_epi32(coeff1, coeff2); |
68 | 0 | } |
69 | | |
70 | | static inline void store_coefficients_avx2(__m256i coeff_vals, |
71 | 0 | tran_low_t *coeff_ptr) { |
72 | 0 | __m256i coeff_sign = _mm256_srai_epi16(coeff_vals, 15); |
73 | 0 | __m256i coeff_vals_lo = _mm256_unpacklo_epi16(coeff_vals, coeff_sign); |
74 | 0 | __m256i coeff_vals_hi = _mm256_unpackhi_epi16(coeff_vals, coeff_sign); |
75 | 0 | _mm256_store_si256((__m256i *)coeff_ptr, coeff_vals_lo); |
76 | 0 | _mm256_store_si256((__m256i *)(coeff_ptr + 8), coeff_vals_hi); |
77 | 0 | } |
78 | | |
79 | 0 | static inline uint16_t quant_gather_eob(__m256i eob) { |
80 | 0 | const __m128i eob_lo = _mm256_castsi256_si128(eob); |
81 | 0 | const __m128i eob_hi = _mm256_extractf128_si256(eob, 1); |
82 | 0 | __m128i eob_s = _mm_max_epi16(eob_lo, eob_hi); |
83 | 0 | eob_s = _mm_subs_epu16(_mm_set1_epi16(INT16_MAX), eob_s); |
84 | 0 | eob_s = _mm_minpos_epu16(eob_s); |
85 | 0 | return INT16_MAX - _mm_extract_epi16(eob_s, 0); |
86 | 0 | } |
87 | | |
88 | 0 | static inline int16_t accumulate_eob256(__m256i eob256) { |
89 | 0 | const __m128i eob_lo = _mm256_castsi256_si128(eob256); |
90 | 0 | const __m128i eob_hi = _mm256_extractf128_si256(eob256, 1); |
91 | 0 | __m128i eob = _mm_max_epi16(eob_lo, eob_hi); |
92 | 0 | __m128i eob_shuffled = _mm_shuffle_epi32(eob, 0xe); |
93 | 0 | eob = _mm_max_epi16(eob, eob_shuffled); |
94 | 0 | eob_shuffled = _mm_shufflelo_epi16(eob, 0xe); |
95 | 0 | eob = _mm_max_epi16(eob, eob_shuffled); |
96 | 0 | eob_shuffled = _mm_shufflelo_epi16(eob, 0x1); |
97 | 0 | eob = _mm_max_epi16(eob, eob_shuffled); |
98 | 0 | return _mm_extract_epi16(eob, 1); |
99 | 0 | } |
100 | | |
101 | | static AOM_FORCE_INLINE void quantize_lp_16_first( |
102 | | const int16_t *coeff_ptr, const int16_t *iscan_ptr, int16_t *qcoeff_ptr, |
103 | | int16_t *dqcoeff_ptr, __m256i *round256, __m256i *quant256, |
104 | 0 | __m256i *dequant256, __m256i *eob) { |
105 | 0 | const __m256i coeff = _mm256_loadu_si256((const __m256i *)coeff_ptr); |
106 | 0 | const __m256i abs_coeff = _mm256_abs_epi16(coeff); |
107 | 0 | const __m256i tmp_rnd = _mm256_adds_epi16(abs_coeff, *round256); |
108 | 0 | const __m256i abs_qcoeff = _mm256_mulhi_epi16(tmp_rnd, *quant256); |
109 | 0 | const __m256i qcoeff = _mm256_sign_epi16(abs_qcoeff, coeff); |
110 | 0 | const __m256i dqcoeff = _mm256_mullo_epi16(qcoeff, *dequant256); |
111 | 0 | const __m256i nz_mask = |
112 | 0 | _mm256_cmpgt_epi16(abs_qcoeff, _mm256_setzero_si256()); |
113 | |
|
114 | 0 | _mm256_storeu_si256((__m256i *)qcoeff_ptr, qcoeff); |
115 | 0 | _mm256_storeu_si256((__m256i *)dqcoeff_ptr, dqcoeff); |
116 | |
|
117 | 0 | const __m256i iscan = _mm256_loadu_si256((const __m256i *)iscan_ptr); |
118 | 0 | const __m256i iscan_plus1 = _mm256_sub_epi16(iscan, nz_mask); |
119 | 0 | const __m256i nz_iscan = _mm256_and_si256(iscan_plus1, nz_mask); |
120 | 0 | *eob = _mm256_max_epi16(*eob, nz_iscan); |
121 | 0 | } |
122 | | |
123 | | static AOM_FORCE_INLINE void quantize_lp_16( |
124 | | const int16_t *coeff_ptr, intptr_t n_coeffs, const int16_t *iscan_ptr, |
125 | | int16_t *qcoeff_ptr, int16_t *dqcoeff_ptr, __m256i *round256, |
126 | 0 | __m256i *quant256, __m256i *dequant256, __m256i *eob) { |
127 | 0 | const __m256i coeff = |
128 | 0 | _mm256_loadu_si256((const __m256i *)(coeff_ptr + n_coeffs)); |
129 | 0 | const __m256i abs_coeff = _mm256_abs_epi16(coeff); |
130 | 0 | const __m256i tmp_rnd = _mm256_adds_epi16(abs_coeff, *round256); |
131 | 0 | const __m256i abs_qcoeff = _mm256_mulhi_epi16(tmp_rnd, *quant256); |
132 | 0 | const __m256i qcoeff = _mm256_sign_epi16(abs_qcoeff, coeff); |
133 | 0 | const __m256i dqcoeff = _mm256_mullo_epi16(qcoeff, *dequant256); |
134 | 0 | const __m256i nz_mask = |
135 | 0 | _mm256_cmpgt_epi16(abs_qcoeff, _mm256_setzero_si256()); |
136 | |
|
137 | 0 | _mm256_storeu_si256((__m256i *)(qcoeff_ptr + n_coeffs), qcoeff); |
138 | 0 | _mm256_storeu_si256((__m256i *)(dqcoeff_ptr + n_coeffs), dqcoeff); |
139 | |
|
140 | 0 | const __m256i iscan = |
141 | 0 | _mm256_loadu_si256((const __m256i *)(iscan_ptr + n_coeffs)); |
142 | 0 | const __m256i iscan_plus1 = _mm256_sub_epi16(iscan, nz_mask); |
143 | 0 | const __m256i nz_iscan = _mm256_and_si256(iscan_plus1, nz_mask); |
144 | 0 | *eob = _mm256_max_epi16(*eob, nz_iscan); |
145 | 0 | } |
146 | | |
147 | | void av1_quantize_lp_avx2(const int16_t *coeff_ptr, intptr_t n_coeffs, |
148 | | const int16_t *round_ptr, const int16_t *quant_ptr, |
149 | | int16_t *qcoeff_ptr, int16_t *dqcoeff_ptr, |
150 | | const int16_t *dequant_ptr, uint16_t *eob_ptr, |
151 | 0 | const int16_t *scan, const int16_t *iscan) { |
152 | 0 | (void)scan; |
153 | 0 | __m256i eob256 = _mm256_setzero_si256(); |
154 | | |
155 | | // Setup global values. |
156 | 0 | __m256i round256 = |
157 | 0 | _mm256_castsi128_si256(_mm_load_si128((const __m128i *)round_ptr)); |
158 | 0 | __m256i quant256 = |
159 | 0 | _mm256_castsi128_si256(_mm_load_si128((const __m128i *)quant_ptr)); |
160 | 0 | __m256i dequant256 = |
161 | 0 | _mm256_castsi128_si256(_mm_load_si128((const __m128i *)dequant_ptr)); |
162 | | |
163 | | // Populate upper AC values. |
164 | 0 | round256 = _mm256_permute4x64_epi64(round256, 0x54); |
165 | 0 | quant256 = _mm256_permute4x64_epi64(quant256, 0x54); |
166 | 0 | dequant256 = _mm256_permute4x64_epi64(dequant256, 0x54); |
167 | | |
168 | | // Process DC and the first 15 AC coeffs. |
169 | 0 | quantize_lp_16_first(coeff_ptr, iscan, qcoeff_ptr, dqcoeff_ptr, &round256, |
170 | 0 | &quant256, &dequant256, &eob256); |
171 | |
|
172 | 0 | if (n_coeffs > 16) { |
173 | | // Overwrite the DC constants with AC constants |
174 | 0 | dequant256 = _mm256_permute2x128_si256(dequant256, dequant256, 0x31); |
175 | 0 | quant256 = _mm256_permute2x128_si256(quant256, quant256, 0x31); |
176 | 0 | round256 = _mm256_permute2x128_si256(round256, round256, 0x31); |
177 | | |
178 | | // AC only loop. |
179 | 0 | for (int idx = 16; idx < n_coeffs; idx += 16) { |
180 | 0 | quantize_lp_16(coeff_ptr, idx, iscan, qcoeff_ptr, dqcoeff_ptr, &round256, |
181 | 0 | &quant256, &dequant256, &eob256); |
182 | 0 | } |
183 | 0 | } |
184 | |
|
185 | 0 | *eob_ptr = accumulate_eob256(eob256); |
186 | 0 | } |
187 | | |
188 | | static AOM_FORCE_INLINE __m256i get_max_lane_eob(const int16_t *iscan, |
189 | | __m256i v_eobmax, |
190 | 0 | __m256i v_mask) { |
191 | 0 | const __m256i v_iscan = _mm256_loadu_si256((const __m256i *)iscan); |
192 | 0 | const __m256i v_iscan_perm = _mm256_permute4x64_epi64(v_iscan, 0xD8); |
193 | 0 | const __m256i v_iscan_plus1 = _mm256_sub_epi16(v_iscan_perm, v_mask); |
194 | 0 | const __m256i v_nz_iscan = _mm256_and_si256(v_iscan_plus1, v_mask); |
195 | 0 | return _mm256_max_epi16(v_eobmax, v_nz_iscan); |
196 | 0 | } |
197 | | |
198 | | static AOM_FORCE_INLINE void quantize_fp_16( |
199 | | const __m256i *thr, const __m256i *qp, const tran_low_t *coeff_ptr, |
200 | | const int16_t *iscan_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, |
201 | 0 | __m256i *eob) { |
202 | 0 | const __m256i coeff = load_coefficients_avx2(coeff_ptr); |
203 | 0 | const __m256i abs_coeff = _mm256_abs_epi16(coeff); |
204 | 0 | const __m256i mask = _mm256_cmpgt_epi16(abs_coeff, *thr); |
205 | 0 | const int nzflag = _mm256_movemask_epi8(mask); |
206 | |
|
207 | 0 | if (nzflag) { |
208 | 0 | const __m256i tmp_rnd = _mm256_adds_epi16(abs_coeff, qp[0]); |
209 | 0 | const __m256i abs_q = _mm256_mulhi_epi16(tmp_rnd, qp[1]); |
210 | 0 | const __m256i q = _mm256_sign_epi16(abs_q, coeff); |
211 | 0 | const __m256i dq = _mm256_mullo_epi16(q, qp[2]); |
212 | 0 | const __m256i nz_mask = _mm256_cmpgt_epi16(abs_q, _mm256_setzero_si256()); |
213 | |
|
214 | 0 | store_coefficients_avx2(q, qcoeff_ptr); |
215 | 0 | store_coefficients_avx2(dq, dqcoeff_ptr); |
216 | |
|
217 | 0 | *eob = get_max_lane_eob(iscan_ptr, *eob, nz_mask); |
218 | 0 | } else { |
219 | 0 | write_zero(qcoeff_ptr); |
220 | 0 | write_zero(dqcoeff_ptr); |
221 | 0 | } |
222 | 0 | } |
223 | | |
224 | | void av1_quantize_fp_avx2(const tran_low_t *coeff_ptr, intptr_t n_coeffs, |
225 | | const int16_t *zbin_ptr, const int16_t *round_ptr, |
226 | | const int16_t *quant_ptr, |
227 | | const int16_t *quant_shift_ptr, |
228 | | tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, |
229 | | const int16_t *dequant_ptr, uint16_t *eob_ptr, |
230 | 0 | const int16_t *scan_ptr, const int16_t *iscan_ptr) { |
231 | 0 | (void)scan_ptr; |
232 | 0 | (void)zbin_ptr; |
233 | 0 | (void)quant_shift_ptr; |
234 | |
|
235 | 0 | const int log_scale = 0; |
236 | 0 | const int step = 16; |
237 | 0 | __m256i qp[3], thr; |
238 | 0 | __m256i eob = _mm256_setzero_si256(); |
239 | |
|
240 | 0 | init_qp(round_ptr, quant_ptr, dequant_ptr, log_scale, &thr, qp); |
241 | |
|
242 | 0 | quantize_fp_16(&thr, qp, coeff_ptr, iscan_ptr, qcoeff_ptr, dqcoeff_ptr, &eob); |
243 | |
|
244 | 0 | coeff_ptr += step; |
245 | 0 | qcoeff_ptr += step; |
246 | 0 | dqcoeff_ptr += step; |
247 | 0 | iscan_ptr += step; |
248 | 0 | n_coeffs -= step; |
249 | |
|
250 | 0 | update_qp(&thr, qp); |
251 | |
|
252 | 0 | while (n_coeffs > 0) { |
253 | 0 | quantize_fp_16(&thr, qp, coeff_ptr, iscan_ptr, qcoeff_ptr, dqcoeff_ptr, |
254 | 0 | &eob); |
255 | |
|
256 | 0 | coeff_ptr += step; |
257 | 0 | qcoeff_ptr += step; |
258 | 0 | dqcoeff_ptr += step; |
259 | 0 | iscan_ptr += step; |
260 | 0 | n_coeffs -= step; |
261 | 0 | } |
262 | 0 | *eob_ptr = quant_gather_eob(eob); |
263 | 0 | } |
264 | | |
265 | | static AOM_FORCE_INLINE void quantize_fp_32x32( |
266 | | const __m256i *thr, const __m256i *qp, const tran_low_t *coeff_ptr, |
267 | | const int16_t *iscan_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, |
268 | 0 | __m256i *eob) { |
269 | 0 | const __m256i coeff = load_coefficients_avx2(coeff_ptr); |
270 | 0 | const __m256i abs_coeff = _mm256_abs_epi16(coeff); |
271 | 0 | const __m256i mask = _mm256_cmpgt_epi16(abs_coeff, *thr); |
272 | 0 | const int nzflag = _mm256_movemask_epi8(mask); |
273 | |
|
274 | 0 | if (nzflag) { |
275 | 0 | const __m256i tmp_rnd = _mm256_adds_epi16(abs_coeff, qp[0]); |
276 | 0 | const __m256i abs_q = _mm256_mulhi_epu16(tmp_rnd, qp[1]); |
277 | 0 | const __m256i q = _mm256_sign_epi16(abs_q, coeff); |
278 | 0 | const __m256i abs_dq = |
279 | 0 | _mm256_srli_epi16(_mm256_mullo_epi16(abs_q, qp[2]), 1); |
280 | 0 | const __m256i nz_mask = _mm256_cmpgt_epi16(abs_q, _mm256_setzero_si256()); |
281 | 0 | const __m256i dq = _mm256_sign_epi16(abs_dq, coeff); |
282 | |
|
283 | 0 | store_coefficients_avx2(q, qcoeff_ptr); |
284 | 0 | store_coefficients_avx2(dq, dqcoeff_ptr); |
285 | |
|
286 | 0 | *eob = get_max_lane_eob(iscan_ptr, *eob, nz_mask); |
287 | 0 | } else { |
288 | 0 | write_zero(qcoeff_ptr); |
289 | 0 | write_zero(dqcoeff_ptr); |
290 | 0 | } |
291 | 0 | } |
292 | | |
293 | | void av1_quantize_fp_32x32_avx2( |
294 | | const tran_low_t *coeff_ptr, intptr_t n_coeffs, const int16_t *zbin_ptr, |
295 | | const int16_t *round_ptr, const int16_t *quant_ptr, |
296 | | const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, |
297 | | tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, |
298 | 0 | const int16_t *scan_ptr, const int16_t *iscan_ptr) { |
299 | 0 | (void)scan_ptr; |
300 | 0 | (void)zbin_ptr; |
301 | 0 | (void)quant_shift_ptr; |
302 | |
|
303 | 0 | const int log_scale = 1; |
304 | 0 | const unsigned int step = 16; |
305 | 0 | __m256i qp[3], thr; |
306 | 0 | __m256i eob = _mm256_setzero_si256(); |
307 | |
|
308 | 0 | init_qp(round_ptr, quant_ptr, dequant_ptr, log_scale, &thr, qp); |
309 | |
|
310 | 0 | quantize_fp_32x32(&thr, qp, coeff_ptr, iscan_ptr, qcoeff_ptr, dqcoeff_ptr, |
311 | 0 | &eob); |
312 | |
|
313 | 0 | coeff_ptr += step; |
314 | 0 | qcoeff_ptr += step; |
315 | 0 | dqcoeff_ptr += step; |
316 | 0 | iscan_ptr += step; |
317 | 0 | n_coeffs -= step; |
318 | |
|
319 | 0 | update_qp(&thr, qp); |
320 | |
|
321 | 0 | while (n_coeffs > 0) { |
322 | 0 | quantize_fp_32x32(&thr, qp, coeff_ptr, iscan_ptr, qcoeff_ptr, dqcoeff_ptr, |
323 | 0 | &eob); |
324 | |
|
325 | 0 | coeff_ptr += step; |
326 | 0 | qcoeff_ptr += step; |
327 | 0 | dqcoeff_ptr += step; |
328 | 0 | iscan_ptr += step; |
329 | 0 | n_coeffs -= step; |
330 | 0 | } |
331 | 0 | *eob_ptr = quant_gather_eob(eob); |
332 | 0 | } |
333 | | |
334 | | static inline void quantize_fp_64x64(const __m256i *thr, const __m256i *qp, |
335 | | const tran_low_t *coeff_ptr, |
336 | | const int16_t *iscan_ptr, |
337 | | tran_low_t *qcoeff_ptr, |
338 | 0 | tran_low_t *dqcoeff_ptr, __m256i *eob) { |
339 | 0 | const __m256i coeff = load_coefficients_avx2(coeff_ptr); |
340 | 0 | const __m256i abs_coeff = _mm256_abs_epi16(coeff); |
341 | 0 | const __m256i mask = _mm256_cmpgt_epi16(abs_coeff, *thr); |
342 | 0 | const int nzflag = _mm256_movemask_epi8(mask); |
343 | |
|
344 | 0 | if (nzflag) { |
345 | 0 | const __m256i tmp_rnd = |
346 | 0 | _mm256_and_si256(_mm256_adds_epi16(abs_coeff, qp[0]), mask); |
347 | 0 | const __m256i qh = _mm256_slli_epi16(_mm256_mulhi_epi16(tmp_rnd, qp[1]), 2); |
348 | 0 | const __m256i ql = |
349 | 0 | _mm256_srli_epi16(_mm256_mullo_epi16(tmp_rnd, qp[1]), 14); |
350 | 0 | const __m256i abs_q = _mm256_or_si256(qh, ql); |
351 | 0 | const __m256i dqh = _mm256_slli_epi16(_mm256_mulhi_epi16(abs_q, qp[2]), 14); |
352 | 0 | const __m256i dql = _mm256_srli_epi16(_mm256_mullo_epi16(abs_q, qp[2]), 2); |
353 | 0 | const __m256i abs_dq = _mm256_or_si256(dqh, dql); |
354 | 0 | const __m256i q = _mm256_sign_epi16(abs_q, coeff); |
355 | 0 | const __m256i dq = _mm256_sign_epi16(abs_dq, coeff); |
356 | | // Check the signed q/dq value here instead of the absolute value. When |
357 | | // dequant equals 4, the dequant threshold (*thr) becomes 0 after being |
358 | | // scaled down by (1 + log_scale). See init_qp(). When *thr is 0 and the |
359 | | // abs_coeff is 0, the nzflag will be set. As a result, the eob will be |
360 | | // incorrectly calculated. The psign instruction corrects the error by |
361 | | // zeroing out q/dq if coeff is zero. |
362 | 0 | const __m256i z_mask = _mm256_cmpeq_epi16(dq, _mm256_setzero_si256()); |
363 | 0 | const __m256i nz_mask = _mm256_cmpeq_epi16(z_mask, _mm256_setzero_si256()); |
364 | |
|
365 | 0 | store_coefficients_avx2(q, qcoeff_ptr); |
366 | 0 | store_coefficients_avx2(dq, dqcoeff_ptr); |
367 | |
|
368 | 0 | *eob = get_max_lane_eob(iscan_ptr, *eob, nz_mask); |
369 | 0 | } else { |
370 | 0 | write_zero(qcoeff_ptr); |
371 | 0 | write_zero(dqcoeff_ptr); |
372 | 0 | } |
373 | 0 | } |
374 | | |
375 | | void av1_quantize_fp_64x64_avx2( |
376 | | const tran_low_t *coeff_ptr, intptr_t n_coeffs, const int16_t *zbin_ptr, |
377 | | const int16_t *round_ptr, const int16_t *quant_ptr, |
378 | | const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, |
379 | | tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, |
380 | 0 | const int16_t *scan_ptr, const int16_t *iscan_ptr) { |
381 | 0 | (void)scan_ptr; |
382 | 0 | (void)zbin_ptr; |
383 | 0 | (void)quant_shift_ptr; |
384 | |
|
385 | 0 | const int log_scale = 2; |
386 | 0 | const unsigned int step = 16; |
387 | 0 | __m256i qp[3], thr; |
388 | 0 | __m256i eob = _mm256_setzero_si256(); |
389 | |
|
390 | 0 | init_qp(round_ptr, quant_ptr, dequant_ptr, log_scale, &thr, qp); |
391 | |
|
392 | 0 | quantize_fp_64x64(&thr, qp, coeff_ptr, iscan_ptr, qcoeff_ptr, dqcoeff_ptr, |
393 | 0 | &eob); |
394 | |
|
395 | 0 | coeff_ptr += step; |
396 | 0 | qcoeff_ptr += step; |
397 | 0 | dqcoeff_ptr += step; |
398 | 0 | iscan_ptr += step; |
399 | 0 | n_coeffs -= step; |
400 | |
|
401 | 0 | update_qp(&thr, qp); |
402 | |
|
403 | 0 | while (n_coeffs > 0) { |
404 | 0 | quantize_fp_64x64(&thr, qp, coeff_ptr, iscan_ptr, qcoeff_ptr, dqcoeff_ptr, |
405 | 0 | &eob); |
406 | |
|
407 | 0 | coeff_ptr += step; |
408 | 0 | qcoeff_ptr += step; |
409 | 0 | dqcoeff_ptr += step; |
410 | 0 | iscan_ptr += step; |
411 | 0 | n_coeffs -= step; |
412 | 0 | } |
413 | 0 | *eob_ptr = quant_gather_eob(eob); |
414 | 0 | } |