/src/aom/aom_dsp/x86/adaptive_quantize_sse2.c
Line | Count | Source |
1 | | /* |
2 | | * Copyright (c) 2019, Alliance for Open Media. All rights reserved. |
3 | | * |
4 | | * This source code is subject to the terms of the BSD 2 Clause License and |
5 | | * the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License |
6 | | * was not distributed with this source code in the LICENSE file, you can |
7 | | * obtain it at www.aomedia.org/license/software. If the Alliance for Open |
8 | | * Media Patent License 1.0 was not distributed with this source code in the |
9 | | * PATENTS file, you can obtain it at www.aomedia.org/license/patent. |
10 | | */ |
11 | | |
12 | | #include <assert.h> |
13 | | #include <emmintrin.h> |
14 | | #include "config/aom_dsp_rtcd.h" |
15 | | #include "aom/aom_integer.h" |
16 | | #include "aom_dsp/quantize.h" |
17 | | #include "aom_dsp/x86/quantize_x86.h" |
18 | | |
19 | | void aom_quantize_b_adaptive_sse2( |
20 | | const tran_low_t *coeff_ptr, intptr_t n_coeffs, const int16_t *zbin_ptr, |
21 | | const int16_t *round_ptr, const int16_t *quant_ptr, |
22 | | const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, |
23 | | tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, |
24 | 0 | const int16_t *scan, const int16_t *iscan) { |
25 | 0 | int index = 16; |
26 | 0 | int non_zero_count = 0; |
27 | 0 | int non_zero_count_prescan_add_zero = 0; |
28 | 0 | int is_found0 = 0, is_found1 = 0; |
29 | 0 | int eob = -1; |
30 | 0 | const __m128i zero = _mm_setzero_si128(); |
31 | 0 | __m128i zbin, round, quant, dequant, shift; |
32 | 0 | __m128i coeff0, coeff1, coeff0_sign, coeff1_sign; |
33 | 0 | __m128i qcoeff0, qcoeff1; |
34 | 0 | __m128i cmp_mask0, cmp_mask1; |
35 | 0 | __m128i all_zero; |
36 | 0 | __m128i mask0 = zero, mask1 = zero; |
37 | |
|
38 | 0 | int prescan_add[2]; |
39 | 0 | int thresh[4]; |
40 | 0 | const qm_val_t wt = (1 << AOM_QM_BITS); |
41 | 0 | for (int i = 0; i < 2; ++i) { |
42 | 0 | prescan_add[i] = ROUND_POWER_OF_TWO(dequant_ptr[i] * EOB_FACTOR, 7); |
43 | 0 | thresh[i] = (zbin_ptr[i] * wt + prescan_add[i]) - 1; |
44 | 0 | } |
45 | 0 | thresh[2] = thresh[3] = thresh[1]; |
46 | 0 | __m128i threshold[2]; |
47 | 0 | threshold[0] = _mm_loadu_si128((__m128i *)&thresh[0]); |
48 | 0 | threshold[1] = _mm_unpackhi_epi64(threshold[0], threshold[0]); |
49 | |
|
50 | 0 | #if SKIP_EOB_FACTOR_ADJUST |
51 | 0 | int first = -1; |
52 | 0 | #endif |
53 | | // Setup global values. |
54 | 0 | load_b_values(zbin_ptr, &zbin, round_ptr, &round, quant_ptr, &quant, |
55 | 0 | dequant_ptr, &dequant, quant_shift_ptr, &shift); |
56 | | |
57 | | // Do DC and first 15 AC. |
58 | 0 | coeff0 = load_coefficients(coeff_ptr); |
59 | 0 | coeff1 = load_coefficients(coeff_ptr + 8); |
60 | | |
61 | | // Poor man's abs(). |
62 | 0 | coeff0_sign = _mm_srai_epi16(coeff0, 15); |
63 | 0 | coeff1_sign = _mm_srai_epi16(coeff1, 15); |
64 | 0 | qcoeff0 = invert_sign_sse2(coeff0, coeff0_sign); |
65 | 0 | qcoeff1 = invert_sign_sse2(coeff1, coeff1_sign); |
66 | |
|
67 | 0 | update_mask0(&qcoeff0, &qcoeff1, threshold, iscan, &is_found0, &mask0); |
68 | |
|
69 | 0 | cmp_mask0 = _mm_cmpgt_epi16(qcoeff0, zbin); |
70 | 0 | zbin = _mm_unpackhi_epi64(zbin, zbin); // Switch DC to AC |
71 | 0 | cmp_mask1 = _mm_cmpgt_epi16(qcoeff1, zbin); |
72 | |
|
73 | 0 | update_mask1(&cmp_mask0, &cmp_mask1, iscan, &is_found1, &mask1); |
74 | |
|
75 | 0 | threshold[0] = threshold[1]; |
76 | 0 | all_zero = _mm_or_si128(cmp_mask0, cmp_mask1); |
77 | 0 | if (_mm_movemask_epi8(all_zero) == 0) { |
78 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr), zero); |
79 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr + 4), zero); |
80 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr + 8), zero); |
81 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr + 12), zero); |
82 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr), zero); |
83 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr + 4), zero); |
84 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr + 8), zero); |
85 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr + 12), zero); |
86 | 0 | round = _mm_unpackhi_epi64(round, round); |
87 | 0 | quant = _mm_unpackhi_epi64(quant, quant); |
88 | 0 | shift = _mm_unpackhi_epi64(shift, shift); |
89 | 0 | dequant = _mm_unpackhi_epi64(dequant, dequant); |
90 | 0 | } else { |
91 | 0 | calculate_qcoeff(&qcoeff0, round, quant, shift); |
92 | |
|
93 | 0 | round = _mm_unpackhi_epi64(round, round); |
94 | 0 | quant = _mm_unpackhi_epi64(quant, quant); |
95 | 0 | shift = _mm_unpackhi_epi64(shift, shift); |
96 | |
|
97 | 0 | calculate_qcoeff(&qcoeff1, round, quant, shift); |
98 | | |
99 | | // Reinsert signs |
100 | 0 | qcoeff0 = invert_sign_sse2(qcoeff0, coeff0_sign); |
101 | 0 | qcoeff1 = invert_sign_sse2(qcoeff1, coeff1_sign); |
102 | | |
103 | | // Mask out zbin threshold coeffs |
104 | 0 | qcoeff0 = _mm_and_si128(qcoeff0, cmp_mask0); |
105 | 0 | qcoeff1 = _mm_and_si128(qcoeff1, cmp_mask1); |
106 | |
|
107 | 0 | store_coefficients(qcoeff0, qcoeff_ptr); |
108 | 0 | store_coefficients(qcoeff1, qcoeff_ptr + 8); |
109 | |
|
110 | 0 | coeff0 = calculate_dqcoeff(qcoeff0, dequant); |
111 | 0 | dequant = _mm_unpackhi_epi64(dequant, dequant); |
112 | 0 | coeff1 = calculate_dqcoeff(qcoeff1, dequant); |
113 | |
|
114 | 0 | store_coefficients(coeff0, dqcoeff_ptr); |
115 | 0 | store_coefficients(coeff1, dqcoeff_ptr + 8); |
116 | 0 | } |
117 | | |
118 | | // AC only loop. |
119 | 0 | while (index < n_coeffs) { |
120 | 0 | coeff0 = load_coefficients(coeff_ptr + index); |
121 | 0 | coeff1 = load_coefficients(coeff_ptr + index + 8); |
122 | |
|
123 | 0 | coeff0_sign = _mm_srai_epi16(coeff0, 15); |
124 | 0 | coeff1_sign = _mm_srai_epi16(coeff1, 15); |
125 | 0 | qcoeff0 = invert_sign_sse2(coeff0, coeff0_sign); |
126 | 0 | qcoeff1 = invert_sign_sse2(coeff1, coeff1_sign); |
127 | |
|
128 | 0 | update_mask0(&qcoeff0, &qcoeff1, threshold, iscan + index, &is_found0, |
129 | 0 | &mask0); |
130 | |
|
131 | 0 | cmp_mask0 = _mm_cmpgt_epi16(qcoeff0, zbin); |
132 | 0 | cmp_mask1 = _mm_cmpgt_epi16(qcoeff1, zbin); |
133 | |
|
134 | 0 | update_mask1(&cmp_mask0, &cmp_mask1, iscan + index, &is_found1, &mask1); |
135 | |
|
136 | 0 | all_zero = _mm_or_si128(cmp_mask0, cmp_mask1); |
137 | 0 | if (_mm_movemask_epi8(all_zero) == 0) { |
138 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr + index), zero); |
139 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr + index + 4), zero); |
140 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr + index + 8), zero); |
141 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr + index + 12), zero); |
142 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr + index), zero); |
143 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr + index + 4), zero); |
144 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr + index + 8), zero); |
145 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr + index + 12), zero); |
146 | 0 | index += 16; |
147 | 0 | continue; |
148 | 0 | } |
149 | 0 | calculate_qcoeff(&qcoeff0, round, quant, shift); |
150 | 0 | calculate_qcoeff(&qcoeff1, round, quant, shift); |
151 | |
|
152 | 0 | qcoeff0 = invert_sign_sse2(qcoeff0, coeff0_sign); |
153 | 0 | qcoeff1 = invert_sign_sse2(qcoeff1, coeff1_sign); |
154 | |
|
155 | 0 | qcoeff0 = _mm_and_si128(qcoeff0, cmp_mask0); |
156 | 0 | qcoeff1 = _mm_and_si128(qcoeff1, cmp_mask1); |
157 | |
|
158 | 0 | store_coefficients(qcoeff0, qcoeff_ptr + index); |
159 | 0 | store_coefficients(qcoeff1, qcoeff_ptr + index + 8); |
160 | |
|
161 | 0 | coeff0 = calculate_dqcoeff(qcoeff0, dequant); |
162 | 0 | coeff1 = calculate_dqcoeff(qcoeff1, dequant); |
163 | |
|
164 | 0 | store_coefficients(coeff0, dqcoeff_ptr + index); |
165 | 0 | store_coefficients(coeff1, dqcoeff_ptr + index + 8); |
166 | |
|
167 | 0 | index += 16; |
168 | 0 | } |
169 | 0 | if (is_found0) non_zero_count = calculate_non_zero_count(mask0); |
170 | 0 | if (is_found1) |
171 | 0 | non_zero_count_prescan_add_zero = calculate_non_zero_count(mask1); |
172 | |
|
173 | 0 | for (int i = non_zero_count_prescan_add_zero - 1; i >= non_zero_count; i--) { |
174 | 0 | const int rc = scan[i]; |
175 | 0 | qcoeff_ptr[rc] = 0; |
176 | 0 | dqcoeff_ptr[rc] = 0; |
177 | 0 | } |
178 | |
|
179 | 0 | for (int i = non_zero_count - 1; i >= 0; i--) { |
180 | 0 | const int rc = scan[i]; |
181 | 0 | if (qcoeff_ptr[rc]) { |
182 | 0 | eob = i; |
183 | 0 | break; |
184 | 0 | } |
185 | 0 | } |
186 | |
|
187 | 0 | *eob_ptr = eob + 1; |
188 | 0 | #if SKIP_EOB_FACTOR_ADJUST |
189 | | // TODO(Aniket): Experiment the following loop with intrinsic by combining |
190 | | // with the quantization loop above |
191 | 0 | for (int i = 0; i < non_zero_count; i++) { |
192 | 0 | const int rc = scan[i]; |
193 | 0 | const int qcoeff = qcoeff_ptr[rc]; |
194 | 0 | if (qcoeff) { |
195 | 0 | first = i; |
196 | 0 | break; |
197 | 0 | } |
198 | 0 | } |
199 | 0 | if ((*eob_ptr - 1) >= 0 && first == (*eob_ptr - 1)) { |
200 | 0 | const int rc = scan[(*eob_ptr - 1)]; |
201 | 0 | if (qcoeff_ptr[rc] == 1 || qcoeff_ptr[rc] == -1) { |
202 | 0 | const int coeff = coeff_ptr[rc] * wt; |
203 | 0 | const int coeff_sign = AOMSIGN(coeff); |
204 | 0 | const int abs_coeff = (coeff ^ coeff_sign) - coeff_sign; |
205 | 0 | const int factor = EOB_FACTOR + SKIP_EOB_FACTOR_ADJUST; |
206 | 0 | const int prescan_add_val = |
207 | 0 | ROUND_POWER_OF_TWO(dequant_ptr[rc != 0] * factor, 7); |
208 | 0 | if (abs_coeff < |
209 | 0 | (zbin_ptr[rc != 0] * (1 << AOM_QM_BITS) + prescan_add_val)) { |
210 | 0 | qcoeff_ptr[rc] = 0; |
211 | 0 | dqcoeff_ptr[rc] = 0; |
212 | 0 | *eob_ptr = 0; |
213 | 0 | } |
214 | 0 | } |
215 | 0 | } |
216 | 0 | #endif |
217 | 0 | } |
218 | | |
219 | | void aom_quantize_b_32x32_adaptive_sse2( |
220 | | const tran_low_t *coeff_ptr, intptr_t n_coeffs, const int16_t *zbin_ptr, |
221 | | const int16_t *round_ptr, const int16_t *quant_ptr, |
222 | | const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, |
223 | | tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, |
224 | 0 | const int16_t *scan, const int16_t *iscan) { |
225 | 0 | int index = 16; |
226 | 0 | const int log_scale = 1; |
227 | 0 | int non_zero_count = 0; |
228 | 0 | int non_zero_count_prescan_add_zero = 0; |
229 | 0 | int is_found0 = 0, is_found1 = 0; |
230 | 0 | int eob = -1; |
231 | 0 | const __m128i zero = _mm_setzero_si128(); |
232 | 0 | const __m128i one = _mm_set1_epi16(1); |
233 | 0 | const __m128i log_scale_vec = _mm_set1_epi16(log_scale); |
234 | 0 | __m128i zbin, round, quant, dequant, shift; |
235 | 0 | __m128i coeff0, coeff1, coeff0_sign, coeff1_sign; |
236 | 0 | __m128i qcoeff0, qcoeff1; |
237 | 0 | __m128i cmp_mask0, cmp_mask1; |
238 | 0 | __m128i all_zero; |
239 | 0 | __m128i mask0 = zero, mask1 = zero; |
240 | |
|
241 | 0 | const int zbins[2] = { ROUND_POWER_OF_TWO(zbin_ptr[0], log_scale), |
242 | 0 | ROUND_POWER_OF_TWO(zbin_ptr[1], log_scale) }; |
243 | 0 | int prescan_add[2]; |
244 | 0 | int thresh[4]; |
245 | 0 | const qm_val_t wt = (1 << AOM_QM_BITS); |
246 | 0 | for (int i = 0; i < 2; ++i) { |
247 | 0 | prescan_add[i] = ROUND_POWER_OF_TWO(dequant_ptr[i] * EOB_FACTOR, 7); |
248 | 0 | thresh[i] = (zbins[i] * wt + prescan_add[i]) - 1; |
249 | 0 | } |
250 | 0 | thresh[2] = thresh[3] = thresh[1]; |
251 | 0 | __m128i threshold[2]; |
252 | 0 | threshold[0] = _mm_loadu_si128((__m128i *)&thresh[0]); |
253 | 0 | threshold[1] = _mm_unpackhi_epi64(threshold[0], threshold[0]); |
254 | |
|
255 | 0 | #if SKIP_EOB_FACTOR_ADJUST |
256 | 0 | int first = -1; |
257 | 0 | #endif |
258 | | // Setup global values. |
259 | 0 | zbin = _mm_load_si128((const __m128i *)zbin_ptr); |
260 | 0 | round = _mm_load_si128((const __m128i *)round_ptr); |
261 | 0 | quant = _mm_load_si128((const __m128i *)quant_ptr); |
262 | 0 | dequant = _mm_load_si128((const __m128i *)dequant_ptr); |
263 | 0 | shift = _mm_load_si128((const __m128i *)quant_shift_ptr); |
264 | | |
265 | | // Shift with rounding. |
266 | 0 | zbin = _mm_add_epi16(zbin, log_scale_vec); |
267 | 0 | round = _mm_add_epi16(round, log_scale_vec); |
268 | 0 | zbin = _mm_srli_epi16(zbin, log_scale); |
269 | 0 | round = _mm_srli_epi16(round, log_scale); |
270 | 0 | zbin = _mm_sub_epi16(zbin, one); |
271 | | |
272 | | // Do DC and first 15 AC. |
273 | 0 | coeff0 = load_coefficients(coeff_ptr); |
274 | 0 | coeff1 = load_coefficients(coeff_ptr + 8); |
275 | |
|
276 | 0 | coeff0_sign = _mm_srai_epi16(coeff0, 15); |
277 | 0 | coeff1_sign = _mm_srai_epi16(coeff1, 15); |
278 | 0 | qcoeff0 = invert_sign_sse2(coeff0, coeff0_sign); |
279 | 0 | qcoeff1 = invert_sign_sse2(coeff1, coeff1_sign); |
280 | |
|
281 | 0 | update_mask0(&qcoeff0, &qcoeff1, threshold, iscan, &is_found0, &mask0); |
282 | |
|
283 | 0 | cmp_mask0 = _mm_cmpgt_epi16(qcoeff0, zbin); |
284 | 0 | zbin = _mm_unpackhi_epi64(zbin, zbin); // Switch DC to AC |
285 | 0 | cmp_mask1 = _mm_cmpgt_epi16(qcoeff1, zbin); |
286 | |
|
287 | 0 | update_mask1(&cmp_mask0, &cmp_mask1, iscan, &is_found1, &mask1); |
288 | |
|
289 | 0 | threshold[0] = threshold[1]; |
290 | 0 | all_zero = _mm_or_si128(cmp_mask0, cmp_mask1); |
291 | 0 | if (_mm_movemask_epi8(all_zero) == 0) { |
292 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr), zero); |
293 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr + 4), zero); |
294 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr + 8), zero); |
295 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr + 12), zero); |
296 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr), zero); |
297 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr + 4), zero); |
298 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr + 8), zero); |
299 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr + 12), zero); |
300 | 0 | round = _mm_unpackhi_epi64(round, round); |
301 | 0 | quant = _mm_unpackhi_epi64(quant, quant); |
302 | 0 | shift = _mm_unpackhi_epi64(shift, shift); |
303 | 0 | dequant = _mm_unpackhi_epi64(dequant, dequant); |
304 | 0 | } else { |
305 | 0 | calculate_qcoeff_log_scale(&qcoeff0, round, quant, &shift, &log_scale); |
306 | 0 | round = _mm_unpackhi_epi64(round, round); |
307 | 0 | quant = _mm_unpackhi_epi64(quant, quant); |
308 | 0 | shift = _mm_unpackhi_epi64(shift, shift); |
309 | 0 | calculate_qcoeff_log_scale(&qcoeff1, round, quant, &shift, &log_scale); |
310 | | |
311 | | // Reinsert signs |
312 | 0 | qcoeff0 = invert_sign_sse2(qcoeff0, coeff0_sign); |
313 | 0 | qcoeff1 = invert_sign_sse2(qcoeff1, coeff1_sign); |
314 | | |
315 | | // Mask out zbin threshold coeffs |
316 | 0 | qcoeff0 = _mm_and_si128(qcoeff0, cmp_mask0); |
317 | 0 | qcoeff1 = _mm_and_si128(qcoeff1, cmp_mask1); |
318 | |
|
319 | 0 | store_coefficients(qcoeff0, qcoeff_ptr); |
320 | 0 | store_coefficients(qcoeff1, qcoeff_ptr + 8); |
321 | |
|
322 | 0 | calculate_dqcoeff_and_store_log_scale(qcoeff0, dequant, zero, dqcoeff_ptr, |
323 | 0 | &log_scale); |
324 | 0 | dequant = _mm_unpackhi_epi64(dequant, dequant); |
325 | 0 | calculate_dqcoeff_and_store_log_scale(qcoeff1, dequant, zero, |
326 | 0 | dqcoeff_ptr + 8, &log_scale); |
327 | 0 | } |
328 | | |
329 | | // AC only loop. |
330 | 0 | while (index < n_coeffs) { |
331 | 0 | coeff0 = load_coefficients(coeff_ptr + index); |
332 | 0 | coeff1 = load_coefficients(coeff_ptr + index + 8); |
333 | |
|
334 | 0 | coeff0_sign = _mm_srai_epi16(coeff0, 15); |
335 | 0 | coeff1_sign = _mm_srai_epi16(coeff1, 15); |
336 | 0 | qcoeff0 = invert_sign_sse2(coeff0, coeff0_sign); |
337 | 0 | qcoeff1 = invert_sign_sse2(coeff1, coeff1_sign); |
338 | |
|
339 | 0 | update_mask0(&qcoeff0, &qcoeff1, threshold, iscan + index, &is_found0, |
340 | 0 | &mask0); |
341 | |
|
342 | 0 | cmp_mask0 = _mm_cmpgt_epi16(qcoeff0, zbin); |
343 | 0 | cmp_mask1 = _mm_cmpgt_epi16(qcoeff1, zbin); |
344 | |
|
345 | 0 | update_mask1(&cmp_mask0, &cmp_mask1, iscan + index, &is_found1, &mask1); |
346 | |
|
347 | 0 | all_zero = _mm_or_si128(cmp_mask0, cmp_mask1); |
348 | 0 | if (_mm_movemask_epi8(all_zero) == 0) { |
349 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr + index), zero); |
350 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr + index + 4), zero); |
351 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr + index + 8), zero); |
352 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr + index + 12), zero); |
353 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr + index), zero); |
354 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr + index + 4), zero); |
355 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr + index + 8), zero); |
356 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr + index + 12), zero); |
357 | 0 | index += 16; |
358 | 0 | continue; |
359 | 0 | } |
360 | 0 | calculate_qcoeff_log_scale(&qcoeff0, round, quant, &shift, &log_scale); |
361 | 0 | calculate_qcoeff_log_scale(&qcoeff1, round, quant, &shift, &log_scale); |
362 | |
|
363 | 0 | qcoeff0 = invert_sign_sse2(qcoeff0, coeff0_sign); |
364 | 0 | qcoeff1 = invert_sign_sse2(qcoeff1, coeff1_sign); |
365 | |
|
366 | 0 | qcoeff0 = _mm_and_si128(qcoeff0, cmp_mask0); |
367 | 0 | qcoeff1 = _mm_and_si128(qcoeff1, cmp_mask1); |
368 | |
|
369 | 0 | store_coefficients(qcoeff0, qcoeff_ptr + index); |
370 | 0 | store_coefficients(qcoeff1, qcoeff_ptr + index + 8); |
371 | |
|
372 | 0 | calculate_dqcoeff_and_store_log_scale(qcoeff0, dequant, zero, |
373 | 0 | dqcoeff_ptr + index, &log_scale); |
374 | 0 | calculate_dqcoeff_and_store_log_scale(qcoeff1, dequant, zero, |
375 | 0 | dqcoeff_ptr + index + 8, &log_scale); |
376 | 0 | index += 16; |
377 | 0 | } |
378 | 0 | if (is_found0) non_zero_count = calculate_non_zero_count(mask0); |
379 | 0 | if (is_found1) |
380 | 0 | non_zero_count_prescan_add_zero = calculate_non_zero_count(mask1); |
381 | |
|
382 | 0 | for (int i = non_zero_count_prescan_add_zero - 1; i >= non_zero_count; i--) { |
383 | 0 | const int rc = scan[i]; |
384 | 0 | qcoeff_ptr[rc] = 0; |
385 | 0 | dqcoeff_ptr[rc] = 0; |
386 | 0 | } |
387 | |
|
388 | 0 | for (int i = non_zero_count - 1; i >= 0; i--) { |
389 | 0 | const int rc = scan[i]; |
390 | 0 | if (qcoeff_ptr[rc]) { |
391 | 0 | eob = i; |
392 | 0 | break; |
393 | 0 | } |
394 | 0 | } |
395 | |
|
396 | 0 | *eob_ptr = eob + 1; |
397 | 0 | #if SKIP_EOB_FACTOR_ADJUST |
398 | | // TODO(Aniket): Experiment the following loop with intrinsic by combining |
399 | | // with the quantization loop above |
400 | 0 | for (int i = 0; i < non_zero_count; i++) { |
401 | 0 | const int rc = scan[i]; |
402 | 0 | const int qcoeff = qcoeff_ptr[rc]; |
403 | 0 | if (qcoeff) { |
404 | 0 | first = i; |
405 | 0 | break; |
406 | 0 | } |
407 | 0 | } |
408 | 0 | if ((*eob_ptr - 1) >= 0 && first == (*eob_ptr - 1)) { |
409 | 0 | const int rc = scan[(*eob_ptr - 1)]; |
410 | 0 | if (qcoeff_ptr[rc] == 1 || qcoeff_ptr[rc] == -1) { |
411 | 0 | const int coeff = coeff_ptr[rc] * wt; |
412 | 0 | const int coeff_sign = AOMSIGN(coeff); |
413 | 0 | const int abs_coeff = (coeff ^ coeff_sign) - coeff_sign; |
414 | 0 | const int factor = EOB_FACTOR + SKIP_EOB_FACTOR_ADJUST; |
415 | 0 | const int prescan_add_val = |
416 | 0 | ROUND_POWER_OF_TWO(dequant_ptr[rc != 0] * factor, 7); |
417 | 0 | if (abs_coeff < (zbins[rc != 0] * (1 << AOM_QM_BITS) + prescan_add_val)) { |
418 | 0 | qcoeff_ptr[rc] = 0; |
419 | 0 | dqcoeff_ptr[rc] = 0; |
420 | 0 | *eob_ptr = 0; |
421 | 0 | } |
422 | 0 | } |
423 | 0 | } |
424 | 0 | #endif |
425 | 0 | } |
426 | | |
427 | | void aom_quantize_b_64x64_adaptive_sse2( |
428 | | const tran_low_t *coeff_ptr, intptr_t n_coeffs, const int16_t *zbin_ptr, |
429 | | const int16_t *round_ptr, const int16_t *quant_ptr, |
430 | | const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, |
431 | | tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, |
432 | 0 | const int16_t *scan, const int16_t *iscan) { |
433 | 0 | int index = 16; |
434 | 0 | const int log_scale = 2; |
435 | 0 | int non_zero_count = 0; |
436 | 0 | int non_zero_count_prescan_add_zero = 0; |
437 | 0 | int is_found0 = 0, is_found1 = 0; |
438 | 0 | int eob = -1; |
439 | 0 | const __m128i zero = _mm_setzero_si128(); |
440 | 0 | const __m128i one = _mm_set1_epi16(1); |
441 | 0 | const __m128i log_scale_vec = _mm_set1_epi16(log_scale); |
442 | 0 | __m128i zbin, round, quant, dequant, shift; |
443 | 0 | __m128i coeff0, coeff1, coeff0_sign, coeff1_sign; |
444 | 0 | __m128i qcoeff0, qcoeff1; |
445 | 0 | __m128i cmp_mask0, cmp_mask1; |
446 | 0 | __m128i all_zero; |
447 | 0 | __m128i mask0 = zero, mask1 = zero; |
448 | |
|
449 | 0 | const int zbins[2] = { ROUND_POWER_OF_TWO(zbin_ptr[0], log_scale), |
450 | 0 | ROUND_POWER_OF_TWO(zbin_ptr[1], log_scale) }; |
451 | 0 | int prescan_add[2]; |
452 | 0 | int thresh[4]; |
453 | 0 | const qm_val_t wt = (1 << AOM_QM_BITS); |
454 | 0 | for (int i = 0; i < 2; ++i) { |
455 | 0 | prescan_add[i] = ROUND_POWER_OF_TWO(dequant_ptr[i] * EOB_FACTOR, 7); |
456 | 0 | thresh[i] = (zbins[i] * wt + prescan_add[i]) - 1; |
457 | 0 | } |
458 | 0 | thresh[2] = thresh[3] = thresh[1]; |
459 | 0 | __m128i threshold[2]; |
460 | 0 | threshold[0] = _mm_loadu_si128((__m128i *)&thresh[0]); |
461 | 0 | threshold[1] = _mm_unpackhi_epi64(threshold[0], threshold[0]); |
462 | |
|
463 | 0 | #if SKIP_EOB_FACTOR_ADJUST |
464 | 0 | int first = -1; |
465 | 0 | #endif |
466 | | // Setup global values. |
467 | 0 | zbin = _mm_load_si128((const __m128i *)zbin_ptr); |
468 | 0 | round = _mm_load_si128((const __m128i *)round_ptr); |
469 | 0 | quant = _mm_load_si128((const __m128i *)quant_ptr); |
470 | 0 | dequant = _mm_load_si128((const __m128i *)dequant_ptr); |
471 | 0 | shift = _mm_load_si128((const __m128i *)quant_shift_ptr); |
472 | | |
473 | | // Shift with rounding. |
474 | 0 | zbin = _mm_add_epi16(zbin, log_scale_vec); |
475 | 0 | round = _mm_add_epi16(round, log_scale_vec); |
476 | 0 | zbin = _mm_srli_epi16(zbin, log_scale); |
477 | 0 | round = _mm_srli_epi16(round, log_scale); |
478 | 0 | zbin = _mm_sub_epi16(zbin, one); |
479 | | |
480 | | // Do DC and first 15 AC. |
481 | 0 | coeff0 = load_coefficients(coeff_ptr); |
482 | 0 | coeff1 = load_coefficients(coeff_ptr + 8); |
483 | |
|
484 | 0 | coeff0_sign = _mm_srai_epi16(coeff0, 15); |
485 | 0 | coeff1_sign = _mm_srai_epi16(coeff1, 15); |
486 | 0 | qcoeff0 = invert_sign_sse2(coeff0, coeff0_sign); |
487 | 0 | qcoeff1 = invert_sign_sse2(coeff1, coeff1_sign); |
488 | |
|
489 | 0 | update_mask0(&qcoeff0, &qcoeff1, threshold, iscan, &is_found0, &mask0); |
490 | |
|
491 | 0 | cmp_mask0 = _mm_cmpgt_epi16(qcoeff0, zbin); |
492 | 0 | zbin = _mm_unpackhi_epi64(zbin, zbin); // Switch DC to AC |
493 | 0 | cmp_mask1 = _mm_cmpgt_epi16(qcoeff1, zbin); |
494 | |
|
495 | 0 | update_mask1(&cmp_mask0, &cmp_mask1, iscan, &is_found1, &mask1); |
496 | |
|
497 | 0 | threshold[0] = threshold[1]; |
498 | 0 | all_zero = _mm_or_si128(cmp_mask0, cmp_mask1); |
499 | 0 | if (_mm_movemask_epi8(all_zero) == 0) { |
500 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr), zero); |
501 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr + 4), zero); |
502 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr + 8), zero); |
503 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr + 12), zero); |
504 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr), zero); |
505 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr + 4), zero); |
506 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr + 8), zero); |
507 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr + 12), zero); |
508 | 0 | round = _mm_unpackhi_epi64(round, round); |
509 | 0 | quant = _mm_unpackhi_epi64(quant, quant); |
510 | 0 | shift = _mm_unpackhi_epi64(shift, shift); |
511 | 0 | dequant = _mm_unpackhi_epi64(dequant, dequant); |
512 | 0 | } else { |
513 | 0 | calculate_qcoeff_log_scale(&qcoeff0, round, quant, &shift, &log_scale); |
514 | 0 | round = _mm_unpackhi_epi64(round, round); |
515 | 0 | quant = _mm_unpackhi_epi64(quant, quant); |
516 | 0 | shift = _mm_unpackhi_epi64(shift, shift); |
517 | 0 | calculate_qcoeff_log_scale(&qcoeff1, round, quant, &shift, &log_scale); |
518 | | |
519 | | // Reinsert signs |
520 | 0 | qcoeff0 = invert_sign_sse2(qcoeff0, coeff0_sign); |
521 | 0 | qcoeff1 = invert_sign_sse2(qcoeff1, coeff1_sign); |
522 | | |
523 | | // Mask out zbin threshold coeffs |
524 | 0 | qcoeff0 = _mm_and_si128(qcoeff0, cmp_mask0); |
525 | 0 | qcoeff1 = _mm_and_si128(qcoeff1, cmp_mask1); |
526 | |
|
527 | 0 | store_coefficients(qcoeff0, qcoeff_ptr); |
528 | 0 | store_coefficients(qcoeff1, qcoeff_ptr + 8); |
529 | |
|
530 | 0 | calculate_dqcoeff_and_store_log_scale(qcoeff0, dequant, zero, dqcoeff_ptr, |
531 | 0 | &log_scale); |
532 | 0 | dequant = _mm_unpackhi_epi64(dequant, dequant); |
533 | 0 | calculate_dqcoeff_and_store_log_scale(qcoeff1, dequant, zero, |
534 | 0 | dqcoeff_ptr + 8, &log_scale); |
535 | 0 | } |
536 | | |
537 | | // AC only loop. |
538 | 0 | while (index < n_coeffs) { |
539 | 0 | coeff0 = load_coefficients(coeff_ptr + index); |
540 | 0 | coeff1 = load_coefficients(coeff_ptr + index + 8); |
541 | |
|
542 | 0 | coeff0_sign = _mm_srai_epi16(coeff0, 15); |
543 | 0 | coeff1_sign = _mm_srai_epi16(coeff1, 15); |
544 | 0 | qcoeff0 = invert_sign_sse2(coeff0, coeff0_sign); |
545 | 0 | qcoeff1 = invert_sign_sse2(coeff1, coeff1_sign); |
546 | |
|
547 | 0 | update_mask0(&qcoeff0, &qcoeff1, threshold, iscan + index, &is_found0, |
548 | 0 | &mask0); |
549 | |
|
550 | 0 | cmp_mask0 = _mm_cmpgt_epi16(qcoeff0, zbin); |
551 | 0 | cmp_mask1 = _mm_cmpgt_epi16(qcoeff1, zbin); |
552 | |
|
553 | 0 | update_mask1(&cmp_mask0, &cmp_mask1, iscan + index, &is_found1, &mask1); |
554 | |
|
555 | 0 | all_zero = _mm_or_si128(cmp_mask0, cmp_mask1); |
556 | 0 | if (_mm_movemask_epi8(all_zero) == 0) { |
557 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr + index), zero); |
558 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr + index + 4), zero); |
559 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr + index + 8), zero); |
560 | 0 | _mm_store_si128((__m128i *)(qcoeff_ptr + index + 12), zero); |
561 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr + index), zero); |
562 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr + index + 4), zero); |
563 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr + index + 8), zero); |
564 | 0 | _mm_store_si128((__m128i *)(dqcoeff_ptr + index + 12), zero); |
565 | 0 | index += 16; |
566 | 0 | continue; |
567 | 0 | } |
568 | 0 | calculate_qcoeff_log_scale(&qcoeff0, round, quant, &shift, &log_scale); |
569 | 0 | calculate_qcoeff_log_scale(&qcoeff1, round, quant, &shift, &log_scale); |
570 | |
|
571 | 0 | qcoeff0 = invert_sign_sse2(qcoeff0, coeff0_sign); |
572 | 0 | qcoeff1 = invert_sign_sse2(qcoeff1, coeff1_sign); |
573 | |
|
574 | 0 | qcoeff0 = _mm_and_si128(qcoeff0, cmp_mask0); |
575 | 0 | qcoeff1 = _mm_and_si128(qcoeff1, cmp_mask1); |
576 | |
|
577 | 0 | store_coefficients(qcoeff0, qcoeff_ptr + index); |
578 | 0 | store_coefficients(qcoeff1, qcoeff_ptr + index + 8); |
579 | |
|
580 | 0 | calculate_dqcoeff_and_store_log_scale(qcoeff0, dequant, zero, |
581 | 0 | dqcoeff_ptr + index, &log_scale); |
582 | 0 | calculate_dqcoeff_and_store_log_scale(qcoeff1, dequant, zero, |
583 | 0 | dqcoeff_ptr + index + 8, &log_scale); |
584 | 0 | index += 16; |
585 | 0 | } |
586 | 0 | if (is_found0) non_zero_count = calculate_non_zero_count(mask0); |
587 | 0 | if (is_found1) |
588 | 0 | non_zero_count_prescan_add_zero = calculate_non_zero_count(mask1); |
589 | |
|
590 | 0 | for (int i = non_zero_count_prescan_add_zero - 1; i >= non_zero_count; i--) { |
591 | 0 | const int rc = scan[i]; |
592 | 0 | qcoeff_ptr[rc] = 0; |
593 | 0 | dqcoeff_ptr[rc] = 0; |
594 | 0 | } |
595 | |
|
596 | 0 | for (int i = non_zero_count - 1; i >= 0; i--) { |
597 | 0 | const int rc = scan[i]; |
598 | 0 | if (qcoeff_ptr[rc]) { |
599 | 0 | eob = i; |
600 | 0 | break; |
601 | 0 | } |
602 | 0 | } |
603 | |
|
604 | 0 | *eob_ptr = eob + 1; |
605 | 0 | #if SKIP_EOB_FACTOR_ADJUST |
606 | | // TODO(Aniket): Experiment the following loop with intrinsic by combining |
607 | | // with the quantization loop above |
608 | 0 | for (int i = 0; i < non_zero_count; i++) { |
609 | 0 | const int rc = scan[i]; |
610 | 0 | const int qcoeff = qcoeff_ptr[rc]; |
611 | 0 | if (qcoeff) { |
612 | 0 | first = i; |
613 | 0 | break; |
614 | 0 | } |
615 | 0 | } |
616 | 0 | if ((*eob_ptr - 1) >= 0 && first == (*eob_ptr - 1)) { |
617 | 0 | const int rc = scan[(*eob_ptr - 1)]; |
618 | 0 | if (qcoeff_ptr[rc] == 1 || qcoeff_ptr[rc] == -1) { |
619 | 0 | const int coeff = coeff_ptr[rc] * wt; |
620 | 0 | const int coeff_sign = AOMSIGN(coeff); |
621 | 0 | const int abs_coeff = (coeff ^ coeff_sign) - coeff_sign; |
622 | 0 | const int factor = EOB_FACTOR + SKIP_EOB_FACTOR_ADJUST; |
623 | 0 | const int prescan_add_val = |
624 | 0 | ROUND_POWER_OF_TWO(dequant_ptr[rc != 0] * factor, 7); |
625 | 0 | if (abs_coeff < (zbins[rc != 0] * (1 << AOM_QM_BITS) + prescan_add_val)) { |
626 | 0 | qcoeff_ptr[rc] = 0; |
627 | 0 | dqcoeff_ptr[rc] = 0; |
628 | 0 | *eob_ptr = 0; |
629 | 0 | } |
630 | 0 | } |
631 | 0 | } |
632 | 0 | #endif |
633 | 0 | } |