Coverage Report

Created: 2026-09-07 06:44

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/aom/av1/encoder/x86/encodetxb_sse2.c
Line
Count
Source
1
/*
2
 * Copyright (c) 2017, Alliance for Open Media. All rights reserved.
3
 *
4
 * This source code is subject to the terms of the BSD 2 Clause License and
5
 * the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
6
 * was not distributed with this source code in the LICENSE file, you can
7
 * obtain it at www.aomedia.org/license/software. If the Alliance for Open
8
 * Media Patent License 1.0 was not distributed with this source code in the
9
 * PATENTS file, you can obtain it at www.aomedia.org/license/patent.
10
 */
11
12
#include <assert.h>
13
#include <emmintrin.h>  // SSE2
14
15
#include "aom/aom_integer.h"
16
#include "aom_dsp/x86/mem_sse2.h"
17
#include "av1/common/av1_common_int.h"
18
#include "av1/common/txb_common.h"
19
20
static inline void load_levels_4x4x5_sse2(const uint8_t *const src,
21
                                          const int stride,
22
                                          const ptrdiff_t *const offsets,
23
0
                                          __m128i *const level) {
24
0
  level[0] = load_8bit_4x4_to_1_reg_sse2(src + 1, stride);
25
0
  level[1] = load_8bit_4x4_to_1_reg_sse2(src + stride, stride);
26
0
  level[2] = load_8bit_4x4_to_1_reg_sse2(src + offsets[0], stride);
27
0
  level[3] = load_8bit_4x4_to_1_reg_sse2(src + offsets[1], stride);
28
0
  level[4] = load_8bit_4x4_to_1_reg_sse2(src + offsets[2], stride);
29
0
}
30
31
static inline void load_levels_8x2x5_sse2(const uint8_t *const src,
32
                                          const int stride,
33
                                          const ptrdiff_t *const offsets,
34
0
                                          __m128i *const level) {
35
0
  level[0] = load_8bit_8x2_to_1_reg_sse2(src + 1, stride);
36
0
  level[1] = load_8bit_8x2_to_1_reg_sse2(src + stride, stride);
37
0
  level[2] = load_8bit_8x2_to_1_reg_sse2(src + offsets[0], stride);
38
0
  level[3] = load_8bit_8x2_to_1_reg_sse2(src + offsets[1], stride);
39
0
  level[4] = load_8bit_8x2_to_1_reg_sse2(src + offsets[2], stride);
40
0
}
41
42
static inline void load_levels_16x1x5_sse2(const uint8_t *const src,
43
                                           const int stride,
44
                                           const ptrdiff_t *const offsets,
45
0
                                           __m128i *const level) {
46
0
  level[0] = _mm_loadu_si128((__m128i *)(src + 1));
47
0
  level[1] = _mm_loadu_si128((__m128i *)(src + stride));
48
0
  level[2] = _mm_loadu_si128((__m128i *)(src + offsets[0]));
49
0
  level[3] = _mm_loadu_si128((__m128i *)(src + offsets[1]));
50
0
  level[4] = _mm_loadu_si128((__m128i *)(src + offsets[2]));
51
0
}
52
53
0
static inline __m128i get_coeff_contexts_kernel_sse2(__m128i *const level) {
54
0
  const __m128i const_3 = _mm_set1_epi8(3);
55
0
  const __m128i const_4 = _mm_set1_epi8(4);
56
0
  __m128i count;
57
58
0
  count = _mm_min_epu8(level[0], const_3);
59
0
  level[1] = _mm_min_epu8(level[1], const_3);
60
0
  level[2] = _mm_min_epu8(level[2], const_3);
61
0
  level[3] = _mm_min_epu8(level[3], const_3);
62
0
  level[4] = _mm_min_epu8(level[4], const_3);
63
0
  count = _mm_add_epi8(count, level[1]);
64
0
  count = _mm_add_epi8(count, level[2]);
65
0
  count = _mm_add_epi8(count, level[3]);
66
0
  count = _mm_add_epi8(count, level[4]);
67
0
  count = _mm_avg_epu8(count, _mm_setzero_si128());
68
0
  count = _mm_min_epu8(count, const_4);
69
0
  return count;
70
0
}
71
72
static inline void get_4_nz_map_contexts_2d(const uint8_t *levels,
73
                                            const int width,
74
                                            const ptrdiff_t *const offsets,
75
0
                                            int8_t *const coeff_contexts) {
76
0
  const int stride = 4 + TX_PAD_HOR;
77
0
  const __m128i pos_to_offset_large = _mm_set1_epi8(21);
78
0
  __m128i pos_to_offset =
79
0
      (width == 4)
80
0
          ? _mm_setr_epi8(0, 1, 6, 6, 1, 6, 6, 21, 6, 6, 21, 21, 6, 21, 21, 21)
81
0
          : _mm_setr_epi8(0, 16, 16, 16, 16, 16, 16, 16, 6, 6, 21, 21, 6, 21,
82
0
                          21, 21);
83
0
  __m128i count;
84
0
  __m128i level[5];
85
0
  int8_t *cc = coeff_contexts;
86
0
  int col = width;
87
88
0
  assert(!(width % 4));
89
90
0
  do {
91
0
    load_levels_4x4x5_sse2(levels, stride, offsets, level);
92
0
    count = get_coeff_contexts_kernel_sse2(level);
93
0
    count = _mm_add_epi8(count, pos_to_offset);
94
0
    _mm_store_si128((__m128i *)cc, count);
95
0
    pos_to_offset = pos_to_offset_large;
96
0
    levels += 4 * stride;
97
0
    cc += 16;
98
0
    col -= 4;
99
0
  } while (col);
100
101
0
  coeff_contexts[0] = 0;
102
0
}
103
104
static inline void get_4_nz_map_contexts_ver(const uint8_t *levels,
105
                                             const int width,
106
                                             const ptrdiff_t *const offsets,
107
0
                                             int8_t *coeff_contexts) {
108
0
  const int stride = 4 + TX_PAD_HOR;
109
0
  const __m128i pos_to_offset =
110
0
      _mm_setr_epi8(SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 5,
111
0
                    SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
112
0
                    SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 5,
113
0
                    SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
114
0
                    SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 5,
115
0
                    SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
116
0
                    SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 5,
117
0
                    SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10);
118
0
  __m128i count;
119
0
  __m128i level[5];
120
0
  int col = width;
121
122
0
  assert(!(width % 4));
123
124
0
  do {
125
0
    load_levels_4x4x5_sse2(levels, stride, offsets, level);
126
0
    count = get_coeff_contexts_kernel_sse2(level);
127
0
    count = _mm_add_epi8(count, pos_to_offset);
128
0
    _mm_store_si128((__m128i *)coeff_contexts, count);
129
0
    levels += 4 * stride;
130
0
    coeff_contexts += 16;
131
0
    col -= 4;
132
0
  } while (col);
133
0
}
134
135
static inline void get_4_nz_map_contexts_hor(const uint8_t *levels,
136
                                             const int width,
137
                                             const ptrdiff_t *const offsets,
138
0
                                             int8_t *coeff_contexts) {
139
0
  const int stride = 4 + TX_PAD_HOR;
140
0
  const __m128i pos_to_offset_large = _mm_set1_epi8(SIG_COEF_CONTEXTS_2D + 10);
141
0
  __m128i pos_to_offset =
142
0
      _mm_setr_epi8(SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 0,
143
0
                    SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 0,
144
0
                    SIG_COEF_CONTEXTS_2D + 5, SIG_COEF_CONTEXTS_2D + 5,
145
0
                    SIG_COEF_CONTEXTS_2D + 5, SIG_COEF_CONTEXTS_2D + 5,
146
0
                    SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
147
0
                    SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
148
0
                    SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
149
0
                    SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10);
150
0
  __m128i count;
151
0
  __m128i level[5];
152
0
  int col = width;
153
154
0
  assert(!(width % 4));
155
156
0
  do {
157
0
    load_levels_4x4x5_sse2(levels, stride, offsets, level);
158
0
    count = get_coeff_contexts_kernel_sse2(level);
159
0
    count = _mm_add_epi8(count, pos_to_offset);
160
0
    _mm_store_si128((__m128i *)coeff_contexts, count);
161
0
    pos_to_offset = pos_to_offset_large;
162
0
    levels += 4 * stride;
163
0
    coeff_contexts += 16;
164
0
    col -= 4;
165
0
  } while (col);
166
0
}
167
168
static inline void get_8_coeff_contexts_2d(const uint8_t *levels,
169
                                           const int width,
170
                                           const ptrdiff_t *const offsets,
171
0
                                           int8_t *coeff_contexts) {
172
0
  const int stride = 8 + TX_PAD_HOR;
173
0
  int8_t *cc = coeff_contexts;
174
0
  int col = width;
175
0
  __m128i count;
176
0
  __m128i level[5];
177
0
  __m128i pos_to_offset[3];
178
179
0
  assert(!(width % 2));
180
181
0
  if (width == 8) {
182
0
    pos_to_offset[0] =
183
0
        _mm_setr_epi8(0, 1, 6, 6, 21, 21, 21, 21, 1, 6, 6, 21, 21, 21, 21, 21);
184
0
    pos_to_offset[1] = _mm_setr_epi8(6, 6, 21, 21, 21, 21, 21, 21, 6, 21, 21,
185
0
                                     21, 21, 21, 21, 21);
186
0
  } else if (width < 8) {
187
0
    pos_to_offset[0] = _mm_setr_epi8(0, 11, 6, 6, 21, 21, 21, 21, 11, 11, 6, 21,
188
0
                                     21, 21, 21, 21);
189
0
    pos_to_offset[1] = _mm_setr_epi8(11, 11, 21, 21, 21, 21, 21, 21, 11, 11, 21,
190
0
                                     21, 21, 21, 21, 21);
191
0
  } else {
192
0
    pos_to_offset[0] = _mm_setr_epi8(0, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16,
193
0
                                     16, 16, 16, 16, 16);
194
0
    pos_to_offset[1] = _mm_setr_epi8(6, 6, 21, 21, 21, 21, 21, 21, 6, 21, 21,
195
0
                                     21, 21, 21, 21, 21);
196
0
  }
197
0
  pos_to_offset[2] = _mm_set1_epi8(21);
198
199
0
  do {
200
0
    load_levels_8x2x5_sse2(levels, stride, offsets, level);
201
0
    count = get_coeff_contexts_kernel_sse2(level);
202
0
    count = _mm_add_epi8(count, pos_to_offset[0]);
203
0
    _mm_store_si128((__m128i *)cc, count);
204
0
    pos_to_offset[0] = pos_to_offset[1];
205
0
    pos_to_offset[1] = pos_to_offset[2];
206
0
    levels += 2 * stride;
207
0
    cc += 16;
208
0
    col -= 2;
209
0
  } while (col);
210
211
0
  coeff_contexts[0] = 0;
212
0
}
213
214
static inline void get_8_coeff_contexts_ver(const uint8_t *levels,
215
                                            const int width,
216
                                            const ptrdiff_t *const offsets,
217
0
                                            int8_t *coeff_contexts) {
218
0
  const int stride = 8 + TX_PAD_HOR;
219
0
  const __m128i pos_to_offset =
220
0
      _mm_setr_epi8(SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 5,
221
0
                    SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
222
0
                    SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
223
0
                    SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
224
0
                    SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 5,
225
0
                    SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
226
0
                    SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
227
0
                    SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10);
228
0
  int col = width;
229
0
  __m128i count;
230
0
  __m128i level[5];
231
232
0
  assert(!(width % 2));
233
234
0
  do {
235
0
    load_levels_8x2x5_sse2(levels, stride, offsets, level);
236
0
    count = get_coeff_contexts_kernel_sse2(level);
237
0
    count = _mm_add_epi8(count, pos_to_offset);
238
0
    _mm_store_si128((__m128i *)coeff_contexts, count);
239
0
    levels += 2 * stride;
240
0
    coeff_contexts += 16;
241
0
    col -= 2;
242
0
  } while (col);
243
0
}
244
245
static inline void get_8_coeff_contexts_hor(const uint8_t *levels,
246
                                            const int width,
247
                                            const ptrdiff_t *const offsets,
248
0
                                            int8_t *coeff_contexts) {
249
0
  const int stride = 8 + TX_PAD_HOR;
250
0
  const __m128i pos_to_offset_large = _mm_set1_epi8(SIG_COEF_CONTEXTS_2D + 10);
251
0
  __m128i pos_to_offset =
252
0
      _mm_setr_epi8(SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 0,
253
0
                    SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 0,
254
0
                    SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 0,
255
0
                    SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 0,
256
0
                    SIG_COEF_CONTEXTS_2D + 5, SIG_COEF_CONTEXTS_2D + 5,
257
0
                    SIG_COEF_CONTEXTS_2D + 5, SIG_COEF_CONTEXTS_2D + 5,
258
0
                    SIG_COEF_CONTEXTS_2D + 5, SIG_COEF_CONTEXTS_2D + 5,
259
0
                    SIG_COEF_CONTEXTS_2D + 5, SIG_COEF_CONTEXTS_2D + 5);
260
0
  int col = width;
261
0
  __m128i count;
262
0
  __m128i level[5];
263
264
0
  assert(!(width % 2));
265
266
0
  do {
267
0
    load_levels_8x2x5_sse2(levels, stride, offsets, level);
268
0
    count = get_coeff_contexts_kernel_sse2(level);
269
0
    count = _mm_add_epi8(count, pos_to_offset);
270
0
    _mm_store_si128((__m128i *)coeff_contexts, count);
271
0
    pos_to_offset = pos_to_offset_large;
272
0
    levels += 2 * stride;
273
0
    coeff_contexts += 16;
274
0
    col -= 2;
275
0
  } while (col);
276
0
}
277
278
static inline void get_16n_coeff_contexts_2d(const uint8_t *levels,
279
                                             const int real_width,
280
                                             const int real_height,
281
                                             const int width, const int height,
282
                                             const ptrdiff_t *const offsets,
283
0
                                             int8_t *coeff_contexts) {
284
0
  const int stride = height + TX_PAD_HOR;
285
0
  int8_t *cc = coeff_contexts;
286
0
  int col = width;
287
0
  __m128i pos_to_offset[5];
288
0
  __m128i pos_to_offset_large[3];
289
0
  __m128i count;
290
0
  __m128i level[5];
291
292
0
  assert(!(height % 16));
293
294
0
  pos_to_offset_large[2] = _mm_set1_epi8(21);
295
0
  if (real_width == real_height) {
296
0
    pos_to_offset[0] = _mm_setr_epi8(0, 1, 6, 6, 21, 21, 21, 21, 21, 21, 21, 21,
297
0
                                     21, 21, 21, 21);
298
0
    pos_to_offset[1] = _mm_setr_epi8(1, 6, 6, 21, 21, 21, 21, 21, 21, 21, 21,
299
0
                                     21, 21, 21, 21, 21);
300
0
    pos_to_offset[2] = _mm_setr_epi8(6, 6, 21, 21, 21, 21, 21, 21, 21, 21, 21,
301
0
                                     21, 21, 21, 21, 21);
302
0
    pos_to_offset[3] = _mm_setr_epi8(6, 21, 21, 21, 21, 21, 21, 21, 21, 21, 21,
303
0
                                     21, 21, 21, 21, 21);
304
0
    pos_to_offset[4] = pos_to_offset_large[0] = pos_to_offset_large[1] =
305
0
        pos_to_offset_large[2];
306
0
  } else if (real_width < real_height) {
307
0
    pos_to_offset[0] = _mm_setr_epi8(0, 11, 6, 6, 21, 21, 21, 21, 21, 21, 21,
308
0
                                     21, 21, 21, 21, 21);
309
0
    pos_to_offset[1] = _mm_setr_epi8(11, 11, 6, 21, 21, 21, 21, 21, 21, 21, 21,
310
0
                                     21, 21, 21, 21, 21);
311
0
    pos_to_offset[2] = pos_to_offset[3] = pos_to_offset[4] = _mm_setr_epi8(
312
0
        11, 11, 21, 21, 21, 21, 21, 21, 21, 21, 21, 21, 21, 21, 21, 21);
313
0
    pos_to_offset_large[0] = pos_to_offset_large[1] = pos_to_offset_large[2];
314
0
  } else {  // real_width > real_height
315
0
    pos_to_offset[0] = pos_to_offset[1] = _mm_setr_epi8(
316
0
        16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16, 16);
317
0
    pos_to_offset[2] = _mm_setr_epi8(6, 6, 21, 21, 21, 21, 21, 21, 21, 21, 21,
318
0
                                     21, 21, 21, 21, 21);
319
0
    pos_to_offset[3] = _mm_setr_epi8(6, 21, 21, 21, 21, 21, 21, 21, 21, 21, 21,
320
0
                                     21, 21, 21, 21, 21);
321
0
    pos_to_offset[4] = pos_to_offset_large[2];
322
0
    pos_to_offset_large[0] = pos_to_offset_large[1] = _mm_set1_epi8(16);
323
0
  }
324
325
0
  do {
326
0
    int h = height;
327
328
0
    do {
329
0
      load_levels_16x1x5_sse2(levels, stride, offsets, level);
330
0
      count = get_coeff_contexts_kernel_sse2(level);
331
0
      count = _mm_add_epi8(count, pos_to_offset[0]);
332
0
      _mm_store_si128((__m128i *)cc, count);
333
0
      levels += 16;
334
0
      cc += 16;
335
0
      h -= 16;
336
0
      pos_to_offset[0] = pos_to_offset_large[0];
337
0
    } while (h);
338
339
0
    pos_to_offset[0] = pos_to_offset[1];
340
0
    pos_to_offset[1] = pos_to_offset[2];
341
0
    pos_to_offset[2] = pos_to_offset[3];
342
0
    pos_to_offset[3] = pos_to_offset[4];
343
0
    pos_to_offset_large[0] = pos_to_offset_large[1];
344
0
    pos_to_offset_large[1] = pos_to_offset_large[2];
345
0
    levels += TX_PAD_HOR;
346
0
  } while (--col);
347
348
0
  coeff_contexts[0] = 0;
349
0
}
350
351
static inline void get_16n_coeff_contexts_ver(const uint8_t *levels,
352
                                              const int width, const int height,
353
                                              const ptrdiff_t *const offsets,
354
0
                                              int8_t *coeff_contexts) {
355
0
  const int stride = height + TX_PAD_HOR;
356
0
  const __m128i pos_to_offset_large =
357
0
      _mm_setr_epi8(SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
358
0
                    SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
359
0
                    SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
360
0
                    SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
361
0
                    SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
362
0
                    SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
363
0
                    SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
364
0
                    SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10);
365
0
  __m128i count;
366
0
  __m128i level[5];
367
0
  int col = width;
368
369
0
  assert(!(height % 16));
370
371
0
  do {
372
0
    __m128i pos_to_offset =
373
0
        _mm_setr_epi8(SIG_COEF_CONTEXTS_2D + 0, SIG_COEF_CONTEXTS_2D + 5,
374
0
                      SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
375
0
                      SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
376
0
                      SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
377
0
                      SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
378
0
                      SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
379
0
                      SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10,
380
0
                      SIG_COEF_CONTEXTS_2D + 10, SIG_COEF_CONTEXTS_2D + 10);
381
0
    int h = height;
382
383
0
    do {
384
0
      load_levels_16x1x5_sse2(levels, stride, offsets, level);
385
0
      count = get_coeff_contexts_kernel_sse2(level);
386
0
      count = _mm_add_epi8(count, pos_to_offset);
387
0
      _mm_store_si128((__m128i *)coeff_contexts, count);
388
0
      pos_to_offset = pos_to_offset_large;
389
0
      levels += 16;
390
0
      coeff_contexts += 16;
391
0
      h -= 16;
392
0
    } while (h);
393
394
0
    levels += TX_PAD_HOR;
395
0
  } while (--col);
396
0
}
397
398
static inline void get_16n_coeff_contexts_hor(const uint8_t *levels,
399
                                              const int width, const int height,
400
                                              const ptrdiff_t *const offsets,
401
0
                                              int8_t *coeff_contexts) {
402
0
  const int stride = height + TX_PAD_HOR;
403
0
  __m128i pos_to_offset[3];
404
0
  __m128i count;
405
0
  __m128i level[5];
406
0
  int col = width;
407
408
0
  assert(!(height % 16));
409
410
0
  pos_to_offset[0] = _mm_set1_epi8(SIG_COEF_CONTEXTS_2D + 0);
411
0
  pos_to_offset[1] = _mm_set1_epi8(SIG_COEF_CONTEXTS_2D + 5);
412
0
  pos_to_offset[2] = _mm_set1_epi8(SIG_COEF_CONTEXTS_2D + 10);
413
414
0
  do {
415
0
    int h = height;
416
417
0
    do {
418
0
      load_levels_16x1x5_sse2(levels, stride, offsets, level);
419
0
      count = get_coeff_contexts_kernel_sse2(level);
420
0
      count = _mm_add_epi8(count, pos_to_offset[0]);
421
0
      _mm_store_si128((__m128i *)coeff_contexts, count);
422
0
      levels += 16;
423
0
      coeff_contexts += 16;
424
0
      h -= 16;
425
0
    } while (h);
426
427
0
    pos_to_offset[0] = pos_to_offset[1];
428
0
    pos_to_offset[1] = pos_to_offset[2];
429
0
    levels += TX_PAD_HOR;
430
0
  } while (--col);
431
0
}
432
433
// Note: levels[] must be in the range [0, 127], inclusive.
434
void av1_get_nz_map_contexts_sse2(const uint8_t *const levels,
435
                                  const int16_t *const scan, const int eob,
436
                                  const TX_SIZE tx_size,
437
                                  const TX_CLASS tx_class,
438
0
                                  int8_t *const coeff_contexts) {
439
0
  const int last_idx = eob - 1;
440
0
  if (!last_idx) {
441
0
    coeff_contexts[0] = 0;
442
0
    return;
443
0
  }
444
445
0
  const int real_width = tx_size_wide[tx_size];
446
0
  const int real_height = tx_size_high[tx_size];
447
0
  const int width = get_txb_wide(tx_size);
448
0
  const int height = get_txb_high(tx_size);
449
0
  const int stride = height + TX_PAD_HOR;
450
0
  ptrdiff_t offsets[3];
451
452
  /* coeff_contexts must be 16 byte aligned. */
453
0
  assert(!((intptr_t)coeff_contexts & 0xf));
454
455
0
  if (tx_class == TX_CLASS_2D) {
456
0
    offsets[0] = 0 * stride + 2;
457
0
    offsets[1] = 1 * stride + 1;
458
0
    offsets[2] = 2 * stride + 0;
459
460
0
    if (height == 4) {
461
0
      get_4_nz_map_contexts_2d(levels, width, offsets, coeff_contexts);
462
0
    } else if (height == 8) {
463
0
      get_8_coeff_contexts_2d(levels, width, offsets, coeff_contexts);
464
0
    } else if (height == 16) {
465
0
      get_16n_coeff_contexts_2d(levels, real_width, real_height, width, height,
466
0
                                offsets, coeff_contexts);
467
0
    } else {
468
0
      get_16n_coeff_contexts_2d(levels, real_width, real_height, width, height,
469
0
                                offsets, coeff_contexts);
470
0
    }
471
0
  } else if (tx_class == TX_CLASS_HORIZ) {
472
0
    offsets[0] = 2 * stride;
473
0
    offsets[1] = 3 * stride;
474
0
    offsets[2] = 4 * stride;
475
0
    if (height == 4) {
476
0
      get_4_nz_map_contexts_hor(levels, width, offsets, coeff_contexts);
477
0
    } else if (height == 8) {
478
0
      get_8_coeff_contexts_hor(levels, width, offsets, coeff_contexts);
479
0
    } else {
480
0
      get_16n_coeff_contexts_hor(levels, width, height, offsets,
481
0
                                 coeff_contexts);
482
0
    }
483
0
  } else {  // TX_CLASS_VERT
484
0
    offsets[0] = 2;
485
0
    offsets[1] = 3;
486
0
    offsets[2] = 4;
487
0
    if (height == 4) {
488
0
      get_4_nz_map_contexts_ver(levels, width, offsets, coeff_contexts);
489
0
    } else if (height == 8) {
490
0
      get_8_coeff_contexts_ver(levels, width, offsets, coeff_contexts);
491
0
    } else {
492
0
      get_16n_coeff_contexts_ver(levels, width, height, offsets,
493
0
                                 coeff_contexts);
494
0
    }
495
0
  }
496
497
0
  const int bhl = get_txb_bhl(tx_size);
498
0
  const int pos = scan[last_idx];
499
0
  if (last_idx <= (width << bhl) / 8)
500
0
    coeff_contexts[pos] = 1;
501
0
  else if (last_idx <= (width << bhl) / 4)
502
0
    coeff_contexts[pos] = 2;
503
0
  else
504
0
    coeff_contexts[pos] = 3;
505
0
}