/src/aom/av1/encoder/x86/encodetxb_sse4.c
Line | Count | Source |
1 | | /* |
2 | | * Copyright (c) 2017, Alliance for Open Media. All rights reserved. |
3 | | * |
4 | | * This source code is subject to the terms of the BSD 2 Clause License and |
5 | | * the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License |
6 | | * was not distributed with this source code in the LICENSE file, you can |
7 | | * obtain it at www.aomedia.org/license/software. If the Alliance for Open |
8 | | * Media Patent License 1.0 was not distributed with this source code in the |
9 | | * PATENTS file, you can obtain it at www.aomedia.org/license/patent. |
10 | | */ |
11 | | |
12 | | #include <assert.h> |
13 | | #include <emmintrin.h> // SSE2 |
14 | | #include <smmintrin.h> /* SSE4.1 */ |
15 | | |
16 | | #include "aom/aom_integer.h" |
17 | | #include "av1/common/av1_common_int.h" |
18 | | #include "av1/common/txb_common.h" |
19 | | #include "aom_dsp/x86/synonyms.h" |
20 | | |
21 | | void av1_txb_init_levels_sse4_1(const tran_low_t *const coeff, const int width, |
22 | 0 | const int height, uint8_t *const levels) { |
23 | 0 | const int stride = height + TX_PAD_HOR; |
24 | 0 | const __m128i zeros = _mm_setzero_si128(); |
25 | |
|
26 | 0 | const int32_t bottom_len = sizeof(*levels) * (TX_PAD_BOTTOM * stride); |
27 | 0 | uint8_t *bottom_buf = levels + stride * width; |
28 | 0 | uint8_t *bottom_buf_end = bottom_buf + bottom_len; |
29 | 0 | do { |
30 | 0 | _mm_storeu_si128((__m128i *)(bottom_buf), zeros); |
31 | 0 | bottom_buf += 16; |
32 | 0 | } while (bottom_buf < bottom_buf_end); |
33 | |
|
34 | 0 | int i = 0; |
35 | 0 | uint8_t *ls = levels; |
36 | 0 | const tran_low_t *cf = coeff; |
37 | 0 | if (height == 4) { |
38 | 0 | do { |
39 | 0 | const __m128i coeffA = xx_loadu_128(cf); |
40 | 0 | const __m128i coeffB = xx_loadu_128(cf + 4); |
41 | 0 | const __m128i coeffAB = _mm_packs_epi32(coeffA, coeffB); |
42 | 0 | const __m128i absAB = _mm_abs_epi16(coeffAB); |
43 | 0 | const __m128i absAB8 = _mm_packs_epi16(absAB, zeros); |
44 | 0 | const __m128i lsAB = _mm_unpacklo_epi32(absAB8, zeros); |
45 | 0 | xx_storeu_128(ls, lsAB); |
46 | 0 | ls += (stride << 1); |
47 | 0 | cf += (height << 1); |
48 | 0 | i += 2; |
49 | 0 | } while (i < width); |
50 | 0 | } else if (height == 8) { |
51 | 0 | do { |
52 | 0 | const __m128i coeffA = xx_loadu_128(cf); |
53 | 0 | const __m128i coeffB = xx_loadu_128(cf + 4); |
54 | 0 | const __m128i coeffAB = _mm_packs_epi32(coeffA, coeffB); |
55 | 0 | const __m128i absAB = _mm_abs_epi16(coeffAB); |
56 | 0 | const __m128i absAB8 = _mm_packs_epi16(absAB, zeros); |
57 | 0 | xx_storeu_128(ls, absAB8); |
58 | 0 | ls += stride; |
59 | 0 | cf += height; |
60 | 0 | i += 1; |
61 | 0 | } while (i < width); |
62 | 0 | } else { |
63 | 0 | do { |
64 | 0 | int j = 0; |
65 | 0 | do { |
66 | 0 | const __m128i coeffA = xx_loadu_128(cf); |
67 | 0 | const __m128i coeffB = xx_loadu_128(cf + 4); |
68 | 0 | const __m128i coeffC = xx_loadu_128(cf + 8); |
69 | 0 | const __m128i coeffD = xx_loadu_128(cf + 12); |
70 | 0 | const __m128i coeffAB = _mm_packs_epi32(coeffA, coeffB); |
71 | 0 | const __m128i coeffCD = _mm_packs_epi32(coeffC, coeffD); |
72 | 0 | const __m128i absAB = _mm_abs_epi16(coeffAB); |
73 | 0 | const __m128i absCD = _mm_abs_epi16(coeffCD); |
74 | 0 | const __m128i absABCD = _mm_packs_epi16(absAB, absCD); |
75 | 0 | xx_storeu_128(ls + j, absABCD); |
76 | 0 | j += 16; |
77 | 0 | cf += 16; |
78 | 0 | } while (j < height); |
79 | 0 | *(int32_t *)(ls + height) = 0; |
80 | 0 | ls += stride; |
81 | 0 | i += 1; |
82 | 0 | } while (i < width); |
83 | 0 | } |
84 | 0 | } |