/src/aom/av1/encoder/x86/encodetxb_sse2.c
Line | Count | Source |
1 | | /* |
2 | | * Copyright (c) 2017, Alliance for Open Media. All rights reserved. |
3 | | * |
4 | | * This source code is subject to the terms of the BSD 2 Clause License and |
5 | | * the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License |
6 | | * was not distributed with this source code in the LICENSE file, you can |
7 | | * obtain it at www.aomedia.org/license/software. If the Alliance for Open |
8 | | * Media Patent License 1.0 was not distributed with this source code in the |
9 | | * PATENTS file, you can obtain it at www.aomedia.org/license/patent. |
10 | | */ |
11 | | |
12 | | #include <assert.h> |
13 | | #include <emmintrin.h> // SSE2 |
14 | | |
15 | | #include "aom/aom_integer.h" |
16 | | #include "aom_dsp/x86/mem_sse2.h" |
17 | | #include "av1/common/av1_common_int.h" |
18 | | #include "av1/common/txb_common.h" |
19 | | |
20 | | static inline void load_levels_4x4x5_sse2(const uint8_t *const src, |
21 | | const int stride, |
22 | | const ptrdiff_t *const offsets, |
23 | 0 | __m128i *const level) { |
24 | 0 | level[0] = load_8bit_4x4_to_1_reg_sse2(src + 1, stride); |
25 | 0 | level[1] = load_8bit_4x4_to_1_reg_sse2(src + stride, stride); |
26 | 0 | level[2] = load_8bit_4x4_to_1_reg_sse2(src + offsets[0], stride); |
27 | 0 | level[3] = load_8bit_4x4_to_1_reg_sse2(src + offsets[1], stride); |
28 | 0 | level[4] = load_8bit_4x4_to_1_reg_sse2(src + offsets[2], stride); |
29 | 0 | } |
30 | | |
31 | | static inline void load_levels_8x2x5_sse2(const uint8_t *const src, |
32 | | const int stride, |
33 | | const ptrdiff_t *const offsets, |
34 | 0 | __m128i *const level) { |
35 | 0 | level[0] = load_8bit_8x2_to_1_reg_sse2(src + 1, stride); |
36 | 0 | level[1] = load_8bit_8x2_to_1_reg_sse2(src + stride, stride); |
37 | 0 | level[2] = load_8bit_8x2_to_1_reg_sse2(src + offsets[0], stride); |
38 | 0 | level[3] = load_8bit_8x2_to_1_reg_sse2(src + offsets[1], stride); |
39 | 0 | level[4] = load_8bit_8x2_to_1_reg_sse2(src + offsets[2], stride); |
40 | 0 | } |
41 | | |
42 | | static inline void load_levels_16x1x5_sse2(const uint8_t *const src, |
43 | | const int stride, |
44 | | const ptrdiff_t *const offsets, |
45 | 0 | __m128i *const level) { |
46 | 0 | level[0] = _mm_loadu_si128((__m128i *)(src + 1)); |
47 | 0 | level[1] = _mm_loadu_si128((__m128i *)(src + stride)); |
48 | 0 | level[2] = _mm_loadu_si128((__m128i *)(src + offsets[0])); |
49 | 0 | level[3] = _mm_loadu_si128((__m128i *)(src + offsets[1])); |
50 | 0 | level[4] = _mm_loadu_si128((__m128i *)(src + offsets[2])); |
51 | 0 | } |
52 | | |
53 | 0 | static inline __m128i get_coeff_contexts_kernel_sse2(__m128i *const level) { |
54 | 0 | const __m128i const_3 = _mm_set1_epi8(3); |
55 | 0 | const __m128i const_4 = _mm_set1_epi8(4); |
56 | 0 | __m128i count; |
57 | |
|
58 | 0 | count = _mm_min_epu8(level[0], const_3); |
59 | 0 | level[1] = _mm_min_epu8(level[1], const_3); |
60 | 0 | level[2] = _mm_min_epu8(level[2], const_3); |
61 | 0 | level[3] = _mm_min_epu8(level[3], const_3); |
62 | 0 | level[4] = _mm_min_epu8(level[4], const_3); |
63 | 0 | count = _mm_add_epi8(count, level[1]); |
64 | 0 | count = _mm_add_epi8(count, level[2]); |
65 | 0 | count = _mm_add_epi8(count, level[3]); |
66 | 0 | count = _mm_add_epi8(count, level[4]); |
67 | 0 | count = _mm_avg_epu8(count, _mm_setzero_si128()); |
68 | 0 | count = _mm_min_epu8(count, const_4); |
69 | 0 | return count; |
70 | 0 | } |
71 | | |
72 | | static inline void get_4_nz_map_contexts_2d(const uint8_t *levels, |
73 | | const int width, |
74 | | const ptrdiff_t *const offsets, |
75 | 0 | int8_t *const coeff_contexts) { |
76 | 0 | const int stride = 4 + TX_PAD_HOR; |
77 | 0 | const __m128i pos_to_offset_large = _mm_set1_epi8(21); |
78 | 0 | __m128i pos_to_offset = |
79 | 0 | (width == 4) |
80 | 0 | ? _mm_setr_epi8(0, 1, 6, 6, 1, 6, 6, 21, 6, 6, 21, 21, 6, 21, 21, 21) |
81 | 0 | : _mm_setr_epi8(0, 16, 16, 16, 16, 16, 16, 16, 6, 6, 21, 21, 6, 21, |
82 | 0 | 21, 21); |
83 | 0 | __m128i count; |
84 | 0 | __m128i level[5]; |
85 | 0 | int8_t *cc = coeff_contexts; |
86 | 0 | int col = width; |
87 | |
|
88 | 0 | assert(!(width % 4)); |
89 | | |
90 | 0 | do { |
91 | 0 | load_levels_4x4x5_sse2(levels, stride, offsets, level); |
92 | 0 | count = get_coeff_contexts_kernel_sse2(level); |
93 | 0 | count = _mm_add_epi8(count, pos_to_offset); |
94 | 0 | _mm_store_si128((__m128i *)cc, count); |
95 | 0 | pos_to_offset = pos_to_offset_large; |
96 | 0 | levels += 4 * stride; |
97 | 0 | cc += 16; |
98 | 0 | col -= 4; |
99 | 0 | } while (col); |
100 | |
|
101 | 0 | coeff_contexts[0] = 0; |
102 | 0 | } |
103 | | |
104 | | static inline void get_4_nz_map_contexts_ver(const uint8_t *levels, |
105 | | const int width, |
106 | | const ptrdiff_t *const offsets, |
107 | 0 | int8_t *coeff_contexts) { |
108 | 0 | const int stride = 4 + TX_PAD_HOR; |
109 | 0 | const __m128i pos_to_offset = |
110 | 0 | _mm_setr_epi8(SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 5, |
111 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
112 | 0 | SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 5, |
113 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
114 | 0 | SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 5, |
115 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
116 | 0 | SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 5, |
117 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10); |
118 | 0 | __m128i count; |
119 | 0 | __m128i level[5]; |
120 | 0 | int col = width; |
121 | |
|
122 | 0 | assert(!(width % 4)); |
123 | | |
124 | 0 | do { |
125 | 0 | load_levels_4x4x5_sse2(levels, stride, offsets, level); |
126 | 0 | count = get_coeff_contexts_kernel_sse2(level); |
127 | 0 | count = _mm_add_epi8(count, pos_to_offset); |
128 | 0 | _mm_store_si128((__m128i *)coeff_contexts, count); |
129 | 0 | levels += 4 * stride; |
130 | 0 | coeff_contexts += 16; |
131 | 0 | col -= 4; |
132 | 0 | } while (col); |
133 | 0 | } |
134 | | |
135 | | static inline void get_4_nz_map_contexts_hor(const uint8_t *levels, |
136 | | const int width, |
137 | | const ptrdiff_t *const offsets, |
138 | 0 | int8_t *coeff_contexts) { |
139 | 0 | const int stride = 4 + TX_PAD_HOR; |
140 | 0 | const __m128i pos_to_offset_large = _mm_set1_epi8(SIG_COEF_CONTEXTS_2D + 10); |
141 | 0 | __m128i pos_to_offset = |
142 | 0 | _mm_setr_epi8(SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 0, |
143 | 0 | SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 0, |
144 | 0 | SIG_COEF_CONTEXTS_2D + 5, SIG_COEF_CONTEXTS_2D + 5, |
145 | 0 | SIG_COEF_CONTEXTS_2D + 5, SIG_COEF_CONTEXTS_2D + 5, |
146 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
147 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
148 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
149 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10); |
150 | 0 | __m128i count; |
151 | 0 | __m128i level[5]; |
152 | 0 | int col = width; |
153 | |
|
154 | 0 | assert(!(width % 4)); |
155 | | |
156 | 0 | do { |
157 | 0 | load_levels_4x4x5_sse2(levels, stride, offsets, level); |
158 | 0 | count = get_coeff_contexts_kernel_sse2(level); |
159 | 0 | count = _mm_add_epi8(count, pos_to_offset); |
160 | 0 | _mm_store_si128((__m128i *)coeff_contexts, count); |
161 | 0 | pos_to_offset = pos_to_offset_large; |
162 | 0 | levels += 4 * stride; |
163 | 0 | coeff_contexts += 16; |
164 | 0 | col -= 4; |
165 | 0 | } while (col); |
166 | 0 | } |
167 | | |
168 | | static inline void get_8_coeff_contexts_2d(const uint8_t *levels, |
169 | | const int width, |
170 | | const ptrdiff_t *const offsets, |
171 | 0 | int8_t *coeff_contexts) { |
172 | 0 | const int stride = 8 + TX_PAD_HOR; |
173 | 0 | int8_t *cc = coeff_contexts; |
174 | 0 | int col = width; |
175 | 0 | __m128i count; |
176 | 0 | __m128i level[5]; |
177 | 0 | __m128i pos_to_offset[3]; |
178 | |
|
179 | 0 | assert(!(width % 2)); |
180 | | |
181 | 0 | if (width == 8) { |
182 | 0 | pos_to_offset[0] = |
183 | 0 | _mm_setr_epi8(0, 1, 6, 6, 21, 21, 21, 21, 1, 6, 6, 21, 21, 21, 21, 21); |
184 | 0 | pos_to_offset[1] = _mm_setr_epi8(6, 6, 21, 21, 21, 21, 21, 21, 6, 21, 21, |
185 | 0 | 21, 21, 21, 21, 21); |
186 | 0 | } else if (width < 8) { |
187 | 0 | pos_to_offset[0] = _mm_setr_epi8(0, 11, 6, 6, 21, 21, 21, 21, 11, 11, 6, 21, |
188 | 0 | 21, 21, 21, 21); |
189 | 0 | pos_to_offset[1] = _mm_setr_epi8(11, 11, 21, 21, 21, 21, 21, 21, 11, 11, 21, |
190 | 0 | 21, 21, 21, 21, 21); |
191 | 0 | } else { |
192 | 0 | pos_to_offset[0] = _mm_setr_epi8(0, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, |
193 | 0 | 16, 16, 16, 16, 16); |
194 | 0 | pos_to_offset[1] = _mm_setr_epi8(6, 6, 21, 21, 21, 21, 21, 21, 6, 21, 21, |
195 | 0 | 21, 21, 21, 21, 21); |
196 | 0 | } |
197 | 0 | pos_to_offset[2] = _mm_set1_epi8(21); |
198 | |
|
199 | 0 | do { |
200 | 0 | load_levels_8x2x5_sse2(levels, stride, offsets, level); |
201 | 0 | count = get_coeff_contexts_kernel_sse2(level); |
202 | 0 | count = _mm_add_epi8(count, pos_to_offset[0]); |
203 | 0 | _mm_store_si128((__m128i *)cc, count); |
204 | 0 | pos_to_offset[0] = pos_to_offset[1]; |
205 | 0 | pos_to_offset[1] = pos_to_offset[2]; |
206 | 0 | levels += 2 * stride; |
207 | 0 | cc += 16; |
208 | 0 | col -= 2; |
209 | 0 | } while (col); |
210 | |
|
211 | 0 | coeff_contexts[0] = 0; |
212 | 0 | } |
213 | | |
214 | | static inline void get_8_coeff_contexts_ver(const uint8_t *levels, |
215 | | const int width, |
216 | | const ptrdiff_t *const offsets, |
217 | 0 | int8_t *coeff_contexts) { |
218 | 0 | const int stride = 8 + TX_PAD_HOR; |
219 | 0 | const __m128i pos_to_offset = |
220 | 0 | _mm_setr_epi8(SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 5, |
221 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
222 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
223 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
224 | 0 | SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 5, |
225 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
226 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
227 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10); |
228 | 0 | int col = width; |
229 | 0 | __m128i count; |
230 | 0 | __m128i level[5]; |
231 | |
|
232 | 0 | assert(!(width % 2)); |
233 | | |
234 | 0 | do { |
235 | 0 | load_levels_8x2x5_sse2(levels, stride, offsets, level); |
236 | 0 | count = get_coeff_contexts_kernel_sse2(level); |
237 | 0 | count = _mm_add_epi8(count, pos_to_offset); |
238 | 0 | _mm_store_si128((__m128i *)coeff_contexts, count); |
239 | 0 | levels += 2 * stride; |
240 | 0 | coeff_contexts += 16; |
241 | 0 | col -= 2; |
242 | 0 | } while (col); |
243 | 0 | } |
244 | | |
245 | | static inline void get_8_coeff_contexts_hor(const uint8_t *levels, |
246 | | const int width, |
247 | | const ptrdiff_t *const offsets, |
248 | 0 | int8_t *coeff_contexts) { |
249 | 0 | const int stride = 8 + TX_PAD_HOR; |
250 | 0 | const __m128i pos_to_offset_large = _mm_set1_epi8(SIG_COEF_CONTEXTS_2D + 10); |
251 | 0 | __m128i pos_to_offset = |
252 | 0 | _mm_setr_epi8(SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 0, |
253 | 0 | SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 0, |
254 | 0 | SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 0, |
255 | 0 | SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 0, |
256 | 0 | SIG_COEF_CONTEXTS_2D + 5, SIG_COEF_CONTEXTS_2D + 5, |
257 | 0 | SIG_COEF_CONTEXTS_2D + 5, SIG_COEF_CONTEXTS_2D + 5, |
258 | 0 | SIG_COEF_CONTEXTS_2D + 5, SIG_COEF_CONTEXTS_2D + 5, |
259 | 0 | SIG_COEF_CONTEXTS_2D + 5, SIG_COEF_CONTEXTS_2D + 5); |
260 | 0 | int col = width; |
261 | 0 | __m128i count; |
262 | 0 | __m128i level[5]; |
263 | |
|
264 | 0 | assert(!(width % 2)); |
265 | | |
266 | 0 | do { |
267 | 0 | load_levels_8x2x5_sse2(levels, stride, offsets, level); |
268 | 0 | count = get_coeff_contexts_kernel_sse2(level); |
269 | 0 | count = _mm_add_epi8(count, pos_to_offset); |
270 | 0 | _mm_store_si128((__m128i *)coeff_contexts, count); |
271 | 0 | pos_to_offset = pos_to_offset_large; |
272 | 0 | levels += 2 * stride; |
273 | 0 | coeff_contexts += 16; |
274 | 0 | col -= 2; |
275 | 0 | } while (col); |
276 | 0 | } |
277 | | |
278 | | static inline void get_16n_coeff_contexts_2d(const uint8_t *levels, |
279 | | const int real_width, |
280 | | const int real_height, |
281 | | const int width, const int height, |
282 | | const ptrdiff_t *const offsets, |
283 | 0 | int8_t *coeff_contexts) { |
284 | 0 | const int stride = height + TX_PAD_HOR; |
285 | 0 | int8_t *cc = coeff_contexts; |
286 | 0 | int col = width; |
287 | 0 | __m128i pos_to_offset[5]; |
288 | 0 | __m128i pos_to_offset_large[3]; |
289 | 0 | __m128i count; |
290 | 0 | __m128i level[5]; |
291 | |
|
292 | 0 | assert(!(height % 16)); |
293 | | |
294 | 0 | pos_to_offset_large[2] = _mm_set1_epi8(21); |
295 | 0 | if (real_width == real_height) { |
296 | 0 | pos_to_offset[0] = _mm_setr_epi8(0, 1, 6, 6, 21, 21, 21, 21, 21, 21, 21, 21, |
297 | 0 | 21, 21, 21, 21); |
298 | 0 | pos_to_offset[1] = _mm_setr_epi8(1, 6, 6, 21, 21, 21, 21, 21, 21, 21, 21, |
299 | 0 | 21, 21, 21, 21, 21); |
300 | 0 | pos_to_offset[2] = _mm_setr_epi8(6, 6, 21, 21, 21, 21, 21, 21, 21, 21, 21, |
301 | 0 | 21, 21, 21, 21, 21); |
302 | 0 | pos_to_offset[3] = _mm_setr_epi8(6, 21, 21, 21, 21, 21, 21, 21, 21, 21, 21, |
303 | 0 | 21, 21, 21, 21, 21); |
304 | 0 | pos_to_offset[4] = pos_to_offset_large[0] = pos_to_offset_large[1] = |
305 | 0 | pos_to_offset_large[2]; |
306 | 0 | } else if (real_width < real_height) { |
307 | 0 | pos_to_offset[0] = _mm_setr_epi8(0, 11, 6, 6, 21, 21, 21, 21, 21, 21, 21, |
308 | 0 | 21, 21, 21, 21, 21); |
309 | 0 | pos_to_offset[1] = _mm_setr_epi8(11, 11, 6, 21, 21, 21, 21, 21, 21, 21, 21, |
310 | 0 | 21, 21, 21, 21, 21); |
311 | 0 | pos_to_offset[2] = pos_to_offset[3] = pos_to_offset[4] = _mm_setr_epi8( |
312 | 0 | 11, 11, 21, 21, 21, 21, 21, 21, 21, 21, 21, 21, 21, 21, 21, 21); |
313 | 0 | pos_to_offset_large[0] = pos_to_offset_large[1] = pos_to_offset_large[2]; |
314 | 0 | } else { // real_width > real_height |
315 | 0 | pos_to_offset[0] = pos_to_offset[1] = _mm_setr_epi8( |
316 | 0 | 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16); |
317 | 0 | pos_to_offset[2] = _mm_setr_epi8(6, 6, 21, 21, 21, 21, 21, 21, 21, 21, 21, |
318 | 0 | 21, 21, 21, 21, 21); |
319 | 0 | pos_to_offset[3] = _mm_setr_epi8(6, 21, 21, 21, 21, 21, 21, 21, 21, 21, 21, |
320 | 0 | 21, 21, 21, 21, 21); |
321 | 0 | pos_to_offset[4] = pos_to_offset_large[2]; |
322 | 0 | pos_to_offset_large[0] = pos_to_offset_large[1] = _mm_set1_epi8(16); |
323 | 0 | } |
324 | |
|
325 | 0 | do { |
326 | 0 | int h = height; |
327 | |
|
328 | 0 | do { |
329 | 0 | load_levels_16x1x5_sse2(levels, stride, offsets, level); |
330 | 0 | count = get_coeff_contexts_kernel_sse2(level); |
331 | 0 | count = _mm_add_epi8(count, pos_to_offset[0]); |
332 | 0 | _mm_store_si128((__m128i *)cc, count); |
333 | 0 | levels += 16; |
334 | 0 | cc += 16; |
335 | 0 | h -= 16; |
336 | 0 | pos_to_offset[0] = pos_to_offset_large[0]; |
337 | 0 | } while (h); |
338 | |
|
339 | 0 | pos_to_offset[0] = pos_to_offset[1]; |
340 | 0 | pos_to_offset[1] = pos_to_offset[2]; |
341 | 0 | pos_to_offset[2] = pos_to_offset[3]; |
342 | 0 | pos_to_offset[3] = pos_to_offset[4]; |
343 | 0 | pos_to_offset_large[0] = pos_to_offset_large[1]; |
344 | 0 | pos_to_offset_large[1] = pos_to_offset_large[2]; |
345 | 0 | levels += TX_PAD_HOR; |
346 | 0 | } while (--col); |
347 | |
|
348 | 0 | coeff_contexts[0] = 0; |
349 | 0 | } |
350 | | |
351 | | static inline void get_16n_coeff_contexts_ver(const uint8_t *levels, |
352 | | const int width, const int height, |
353 | | const ptrdiff_t *const offsets, |
354 | 0 | int8_t *coeff_contexts) { |
355 | 0 | const int stride = height + TX_PAD_HOR; |
356 | 0 | const __m128i pos_to_offset_large = |
357 | 0 | _mm_setr_epi8(SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
358 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
359 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
360 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
361 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
362 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
363 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
364 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10); |
365 | 0 | __m128i count; |
366 | 0 | __m128i level[5]; |
367 | 0 | int col = width; |
368 | |
|
369 | 0 | assert(!(height % 16)); |
370 | | |
371 | 0 | do { |
372 | 0 | __m128i pos_to_offset = |
373 | 0 | _mm_setr_epi8(SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 5, |
374 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
375 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
376 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
377 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
378 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
379 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10, |
380 | 0 | SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10); |
381 | 0 | int h = height; |
382 | |
|
383 | 0 | do { |
384 | 0 | load_levels_16x1x5_sse2(levels, stride, offsets, level); |
385 | 0 | count = get_coeff_contexts_kernel_sse2(level); |
386 | 0 | count = _mm_add_epi8(count, pos_to_offset); |
387 | 0 | _mm_store_si128((__m128i *)coeff_contexts, count); |
388 | 0 | pos_to_offset = pos_to_offset_large; |
389 | 0 | levels += 16; |
390 | 0 | coeff_contexts += 16; |
391 | 0 | h -= 16; |
392 | 0 | } while (h); |
393 | |
|
394 | 0 | levels += TX_PAD_HOR; |
395 | 0 | } while (--col); |
396 | 0 | } |
397 | | |
398 | | static inline void get_16n_coeff_contexts_hor(const uint8_t *levels, |
399 | | const int width, const int height, |
400 | | const ptrdiff_t *const offsets, |
401 | 0 | int8_t *coeff_contexts) { |
402 | 0 | const int stride = height + TX_PAD_HOR; |
403 | 0 | __m128i pos_to_offset[3]; |
404 | 0 | __m128i count; |
405 | 0 | __m128i level[5]; |
406 | 0 | int col = width; |
407 | |
|
408 | 0 | assert(!(height % 16)); |
409 | | |
410 | 0 | pos_to_offset[0] = _mm_set1_epi8(SIG_COEF_CONTEXTS_2D + 0); |
411 | 0 | pos_to_offset[1] = _mm_set1_epi8(SIG_COEF_CONTEXTS_2D + 5); |
412 | 0 | pos_to_offset[2] = _mm_set1_epi8(SIG_COEF_CONTEXTS_2D + 10); |
413 | |
|
414 | 0 | do { |
415 | 0 | int h = height; |
416 | |
|
417 | 0 | do { |
418 | 0 | load_levels_16x1x5_sse2(levels, stride, offsets, level); |
419 | 0 | count = get_coeff_contexts_kernel_sse2(level); |
420 | 0 | count = _mm_add_epi8(count, pos_to_offset[0]); |
421 | 0 | _mm_store_si128((__m128i *)coeff_contexts, count); |
422 | 0 | levels += 16; |
423 | 0 | coeff_contexts += 16; |
424 | 0 | h -= 16; |
425 | 0 | } while (h); |
426 | |
|
427 | 0 | pos_to_offset[0] = pos_to_offset[1]; |
428 | 0 | pos_to_offset[1] = pos_to_offset[2]; |
429 | 0 | levels += TX_PAD_HOR; |
430 | 0 | } while (--col); |
431 | 0 | } |
432 | | |
433 | | // Note: levels[] must be in the range [0, 127], inclusive. |
434 | | void av1_get_nz_map_contexts_sse2(const uint8_t *const levels, |
435 | | const int16_t *const scan, const int eob, |
436 | | const TX_SIZE tx_size, |
437 | | const TX_CLASS tx_class, |
438 | 0 | int8_t *const coeff_contexts) { |
439 | 0 | const int last_idx = eob - 1; |
440 | 0 | if (!last_idx) { |
441 | 0 | coeff_contexts[0] = 0; |
442 | 0 | return; |
443 | 0 | } |
444 | | |
445 | 0 | const int real_width = tx_size_wide[tx_size]; |
446 | 0 | const int real_height = tx_size_high[tx_size]; |
447 | 0 | const int width = get_txb_wide(tx_size); |
448 | 0 | const int height = get_txb_high(tx_size); |
449 | 0 | const int stride = height + TX_PAD_HOR; |
450 | 0 | ptrdiff_t offsets[3]; |
451 | | |
452 | | /* coeff_contexts must be 16 byte aligned. */ |
453 | 0 | assert(!((intptr_t)coeff_contexts & 0xf)); |
454 | | |
455 | 0 | if (tx_class == TX_CLASS_2D) { |
456 | 0 | offsets[0] = 0 * stride + 2; |
457 | 0 | offsets[1] = 1 * stride + 1; |
458 | 0 | offsets[2] = 2 * stride + 0; |
459 | |
|
460 | 0 | if (height == 4) { |
461 | 0 | get_4_nz_map_contexts_2d(levels, width, offsets, coeff_contexts); |
462 | 0 | } else if (height == 8) { |
463 | 0 | get_8_coeff_contexts_2d(levels, width, offsets, coeff_contexts); |
464 | 0 | } else if (height == 16) { |
465 | 0 | get_16n_coeff_contexts_2d(levels, real_width, real_height, width, height, |
466 | 0 | offsets, coeff_contexts); |
467 | 0 | } else { |
468 | 0 | get_16n_coeff_contexts_2d(levels, real_width, real_height, width, height, |
469 | 0 | offsets, coeff_contexts); |
470 | 0 | } |
471 | 0 | } else if (tx_class == TX_CLASS_HORIZ) { |
472 | 0 | offsets[0] = 2 * stride; |
473 | 0 | offsets[1] = 3 * stride; |
474 | 0 | offsets[2] = 4 * stride; |
475 | 0 | if (height == 4) { |
476 | 0 | get_4_nz_map_contexts_hor(levels, width, offsets, coeff_contexts); |
477 | 0 | } else if (height == 8) { |
478 | 0 | get_8_coeff_contexts_hor(levels, width, offsets, coeff_contexts); |
479 | 0 | } else { |
480 | 0 | get_16n_coeff_contexts_hor(levels, width, height, offsets, |
481 | 0 | coeff_contexts); |
482 | 0 | } |
483 | 0 | } else { // TX_CLASS_VERT |
484 | 0 | offsets[0] = 2; |
485 | 0 | offsets[1] = 3; |
486 | 0 | offsets[2] = 4; |
487 | 0 | if (height == 4) { |
488 | 0 | get_4_nz_map_contexts_ver(levels, width, offsets, coeff_contexts); |
489 | 0 | } else if (height == 8) { |
490 | 0 | get_8_coeff_contexts_ver(levels, width, offsets, coeff_contexts); |
491 | 0 | } else { |
492 | 0 | get_16n_coeff_contexts_ver(levels, width, height, offsets, |
493 | 0 | coeff_contexts); |
494 | 0 | } |
495 | 0 | } |
496 | |
|
497 | 0 | const int bhl = get_txb_bhl(tx_size); |
498 | 0 | const int pos = scan[last_idx]; |
499 | 0 | if (last_idx <= (width << bhl) / 8) |
500 | 0 | coeff_contexts[pos] = 1; |
501 | 0 | else if (last_idx <= (width << bhl) / 4) |
502 | 0 | coeff_contexts[pos] = 2; |
503 | 0 | else |
504 | 0 | coeff_contexts[pos] = 3; |
505 | 0 | } |