/src/aom/aom_dsp/x86/highbd_variance_sse2.c
Line | Count | Source |
1 | | /* |
2 | | * Copyright (c) 2016, Alliance for Open Media. All rights reserved. |
3 | | * |
4 | | * This source code is subject to the terms of the BSD 2 Clause License and |
5 | | * the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License |
6 | | * was not distributed with this source code in the LICENSE file, you can |
7 | | * obtain it at www.aomedia.org/license/software. If the Alliance for Open |
8 | | * Media Patent License 1.0 was not distributed with this source code in the |
9 | | * PATENTS file, you can obtain it at www.aomedia.org/license/patent. |
10 | | */ |
11 | | |
12 | | #include <assert.h> |
13 | | #include <emmintrin.h> // SSE2 |
14 | | |
15 | | #include "config/aom_config.h" |
16 | | #include "config/aom_dsp_rtcd.h" |
17 | | |
18 | | #include "aom_dsp/x86/synonyms.h" |
19 | | #include "aom_ports/mem.h" |
20 | | |
21 | | #include "av1/common/filter.h" |
22 | | #include "av1/common/reconinter.h" |
23 | | |
24 | | typedef uint32_t (*high_variance_fn_t)(const uint16_t *src, int src_stride, |
25 | | const uint16_t *ref, int ref_stride, |
26 | | uint32_t *sse, int *sum); |
27 | | |
28 | | uint32_t aom_highbd_calc8x8var_sse2(const uint16_t *src, int src_stride, |
29 | | const uint16_t *ref, int ref_stride, |
30 | | uint32_t *sse, int *sum); |
31 | | |
32 | | uint32_t aom_highbd_calc16x16var_sse2(const uint16_t *src, int src_stride, |
33 | | const uint16_t *ref, int ref_stride, |
34 | | uint32_t *sse, int *sum); |
35 | | |
36 | | static void highbd_8_variance_sse2(const uint16_t *src, int src_stride, |
37 | | const uint16_t *ref, int ref_stride, int w, |
38 | | int h, uint32_t *sse, int *sum, |
39 | 0 | high_variance_fn_t var_fn, int block_size) { |
40 | 0 | int i, j; |
41 | |
|
42 | 0 | *sse = 0; |
43 | 0 | *sum = 0; |
44 | |
|
45 | 0 | for (i = 0; i < h; i += block_size) { |
46 | 0 | for (j = 0; j < w; j += block_size) { |
47 | 0 | unsigned int sse0; |
48 | 0 | int sum0; |
49 | 0 | var_fn(src + src_stride * i + j, src_stride, ref + ref_stride * i + j, |
50 | 0 | ref_stride, &sse0, &sum0); |
51 | 0 | *sse += sse0; |
52 | 0 | *sum += sum0; |
53 | 0 | } |
54 | 0 | } |
55 | 0 | } |
56 | | |
57 | | static void highbd_10_variance_sse2(const uint16_t *src, int src_stride, |
58 | | const uint16_t *ref, int ref_stride, int w, |
59 | | int h, uint32_t *sse, int *sum, |
60 | 0 | high_variance_fn_t var_fn, int block_size) { |
61 | 0 | int i, j; |
62 | 0 | uint64_t sse_long = 0; |
63 | 0 | int32_t sum_long = 0; |
64 | |
|
65 | 0 | for (i = 0; i < h; i += block_size) { |
66 | 0 | for (j = 0; j < w; j += block_size) { |
67 | 0 | unsigned int sse0; |
68 | 0 | int sum0; |
69 | 0 | var_fn(src + src_stride * i + j, src_stride, ref + ref_stride * i + j, |
70 | 0 | ref_stride, &sse0, &sum0); |
71 | 0 | sse_long += sse0; |
72 | 0 | sum_long += sum0; |
73 | 0 | } |
74 | 0 | } |
75 | 0 | *sum = ROUND_POWER_OF_TWO(sum_long, 2); |
76 | 0 | *sse = (uint32_t)ROUND_POWER_OF_TWO(sse_long, 4); |
77 | 0 | } |
78 | | |
79 | | static void highbd_12_variance_sse2(const uint16_t *src, int src_stride, |
80 | | const uint16_t *ref, int ref_stride, int w, |
81 | | int h, uint32_t *sse, int *sum, |
82 | 0 | high_variance_fn_t var_fn, int block_size) { |
83 | 0 | int i, j; |
84 | 0 | uint64_t sse_long = 0; |
85 | 0 | int32_t sum_long = 0; |
86 | |
|
87 | 0 | for (i = 0; i < h; i += block_size) { |
88 | 0 | for (j = 0; j < w; j += block_size) { |
89 | 0 | unsigned int sse0; |
90 | 0 | int sum0; |
91 | 0 | var_fn(src + src_stride * i + j, src_stride, ref + ref_stride * i + j, |
92 | 0 | ref_stride, &sse0, &sum0); |
93 | 0 | sse_long += sse0; |
94 | 0 | sum_long += sum0; |
95 | 0 | } |
96 | 0 | } |
97 | 0 | *sum = ROUND_POWER_OF_TWO(sum_long, 4); |
98 | 0 | *sse = (uint32_t)ROUND_POWER_OF_TWO(sse_long, 8); |
99 | 0 | } |
100 | | |
101 | | #define VAR_FN(w, h, block_size, shift) \ |
102 | | uint32_t aom_highbd_8_variance##w##x##h##_sse2( \ |
103 | | const uint8_t *src8, int src_stride, const uint8_t *ref8, \ |
104 | 0 | int ref_stride, uint32_t *sse) { \ |
105 | 0 | int sum; \ |
106 | 0 | uint16_t *src = CONVERT_TO_SHORTPTR(src8); \ |
107 | 0 | uint16_t *ref = CONVERT_TO_SHORTPTR(ref8); \ |
108 | 0 | highbd_8_variance_sse2( \ |
109 | 0 | src, src_stride, ref, ref_stride, w, h, sse, &sum, \ |
110 | 0 | aom_highbd_calc##block_size##x##block_size##var_sse2, block_size); \ |
111 | 0 | return *sse - (uint32_t)(((int64_t)sum * sum) >> shift); \ |
112 | 0 | } \ Unexecuted instantiation: aom_highbd_8_variance128x128_sse2 Unexecuted instantiation: aom_highbd_8_variance128x64_sse2 Unexecuted instantiation: aom_highbd_8_variance64x128_sse2 Unexecuted instantiation: aom_highbd_8_variance64x64_sse2 Unexecuted instantiation: aom_highbd_8_variance64x32_sse2 Unexecuted instantiation: aom_highbd_8_variance32x64_sse2 Unexecuted instantiation: aom_highbd_8_variance32x32_sse2 Unexecuted instantiation: aom_highbd_8_variance32x16_sse2 Unexecuted instantiation: aom_highbd_8_variance16x32_sse2 Unexecuted instantiation: aom_highbd_8_variance16x16_sse2 Unexecuted instantiation: aom_highbd_8_variance16x8_sse2 Unexecuted instantiation: aom_highbd_8_variance8x16_sse2 Unexecuted instantiation: aom_highbd_8_variance8x8_sse2 Unexecuted instantiation: aom_highbd_8_variance8x32_sse2 Unexecuted instantiation: aom_highbd_8_variance32x8_sse2 Unexecuted instantiation: aom_highbd_8_variance16x64_sse2 Unexecuted instantiation: aom_highbd_8_variance64x16_sse2 |
113 | | \ |
114 | | uint32_t aom_highbd_10_variance##w##x##h##_sse2( \ |
115 | | const uint8_t *src8, int src_stride, const uint8_t *ref8, \ |
116 | 0 | int ref_stride, uint32_t *sse) { \ |
117 | 0 | int sum; \ |
118 | 0 | int64_t var; \ |
119 | 0 | uint16_t *src = CONVERT_TO_SHORTPTR(src8); \ |
120 | 0 | uint16_t *ref = CONVERT_TO_SHORTPTR(ref8); \ |
121 | 0 | highbd_10_variance_sse2( \ |
122 | 0 | src, src_stride, ref, ref_stride, w, h, sse, &sum, \ |
123 | 0 | aom_highbd_calc##block_size##x##block_size##var_sse2, block_size); \ |
124 | 0 | var = (int64_t)(*sse) - (((int64_t)sum * sum) >> shift); \ |
125 | 0 | return (var >= 0) ? (uint32_t)var : 0; \ |
126 | 0 | } \ Unexecuted instantiation: aom_highbd_10_variance128x128_sse2 Unexecuted instantiation: aom_highbd_10_variance128x64_sse2 Unexecuted instantiation: aom_highbd_10_variance64x128_sse2 Unexecuted instantiation: aom_highbd_10_variance64x64_sse2 Unexecuted instantiation: aom_highbd_10_variance64x32_sse2 Unexecuted instantiation: aom_highbd_10_variance32x64_sse2 Unexecuted instantiation: aom_highbd_10_variance32x32_sse2 Unexecuted instantiation: aom_highbd_10_variance32x16_sse2 Unexecuted instantiation: aom_highbd_10_variance16x32_sse2 Unexecuted instantiation: aom_highbd_10_variance16x16_sse2 Unexecuted instantiation: aom_highbd_10_variance16x8_sse2 Unexecuted instantiation: aom_highbd_10_variance8x16_sse2 Unexecuted instantiation: aom_highbd_10_variance8x8_sse2 Unexecuted instantiation: aom_highbd_10_variance8x32_sse2 Unexecuted instantiation: aom_highbd_10_variance32x8_sse2 Unexecuted instantiation: aom_highbd_10_variance16x64_sse2 Unexecuted instantiation: aom_highbd_10_variance64x16_sse2 |
127 | | \ |
128 | | uint32_t aom_highbd_12_variance##w##x##h##_sse2( \ |
129 | | const uint8_t *src8, int src_stride, const uint8_t *ref8, \ |
130 | 0 | int ref_stride, uint32_t *sse) { \ |
131 | 0 | int sum; \ |
132 | 0 | int64_t var; \ |
133 | 0 | uint16_t *src = CONVERT_TO_SHORTPTR(src8); \ |
134 | 0 | uint16_t *ref = CONVERT_TO_SHORTPTR(ref8); \ |
135 | 0 | highbd_12_variance_sse2( \ |
136 | 0 | src, src_stride, ref, ref_stride, w, h, sse, &sum, \ |
137 | 0 | aom_highbd_calc##block_size##x##block_size##var_sse2, block_size); \ |
138 | 0 | var = (int64_t)(*sse) - (((int64_t)sum * sum) >> shift); \ |
139 | 0 | return (var >= 0) ? (uint32_t)var : 0; \ |
140 | 0 | } Unexecuted instantiation: aom_highbd_12_variance128x128_sse2 Unexecuted instantiation: aom_highbd_12_variance128x64_sse2 Unexecuted instantiation: aom_highbd_12_variance64x128_sse2 Unexecuted instantiation: aom_highbd_12_variance64x64_sse2 Unexecuted instantiation: aom_highbd_12_variance64x32_sse2 Unexecuted instantiation: aom_highbd_12_variance32x64_sse2 Unexecuted instantiation: aom_highbd_12_variance32x32_sse2 Unexecuted instantiation: aom_highbd_12_variance32x16_sse2 Unexecuted instantiation: aom_highbd_12_variance16x32_sse2 Unexecuted instantiation: aom_highbd_12_variance16x16_sse2 Unexecuted instantiation: aom_highbd_12_variance16x8_sse2 Unexecuted instantiation: aom_highbd_12_variance8x16_sse2 Unexecuted instantiation: aom_highbd_12_variance8x8_sse2 Unexecuted instantiation: aom_highbd_12_variance8x32_sse2 Unexecuted instantiation: aom_highbd_12_variance32x8_sse2 Unexecuted instantiation: aom_highbd_12_variance16x64_sse2 Unexecuted instantiation: aom_highbd_12_variance64x16_sse2 |
141 | | |
142 | | VAR_FN(128, 128, 16, 14) |
143 | | VAR_FN(128, 64, 16, 13) |
144 | | VAR_FN(64, 128, 16, 13) |
145 | | VAR_FN(64, 64, 16, 12) |
146 | | VAR_FN(64, 32, 16, 11) |
147 | | VAR_FN(32, 64, 16, 11) |
148 | | VAR_FN(32, 32, 16, 10) |
149 | | VAR_FN(32, 16, 16, 9) |
150 | | VAR_FN(16, 32, 16, 9) |
151 | | VAR_FN(16, 16, 16, 8) |
152 | | VAR_FN(16, 8, 8, 7) |
153 | | VAR_FN(8, 16, 8, 7) |
154 | | VAR_FN(8, 8, 8, 6) |
155 | | |
156 | | #if !CONFIG_REALTIME_ONLY |
157 | | VAR_FN(8, 32, 8, 8) |
158 | | VAR_FN(32, 8, 8, 8) |
159 | | VAR_FN(16, 64, 16, 10) |
160 | | VAR_FN(64, 16, 16, 10) |
161 | | #endif // !CONFIG_REALTIME_ONLY |
162 | | |
163 | | #undef VAR_FN |
164 | | |
165 | | unsigned int aom_highbd_8_mse16x16_sse2(const uint8_t *src8, int src_stride, |
166 | | const uint8_t *ref8, int ref_stride, |
167 | 0 | unsigned int *sse) { |
168 | 0 | int sum; |
169 | 0 | uint16_t *src = CONVERT_TO_SHORTPTR(src8); |
170 | 0 | uint16_t *ref = CONVERT_TO_SHORTPTR(ref8); |
171 | 0 | highbd_8_variance_sse2(src, src_stride, ref, ref_stride, 16, 16, sse, &sum, |
172 | 0 | aom_highbd_calc16x16var_sse2, 16); |
173 | 0 | return *sse; |
174 | 0 | } |
175 | | |
176 | | unsigned int aom_highbd_10_mse16x16_sse2(const uint8_t *src8, int src_stride, |
177 | | const uint8_t *ref8, int ref_stride, |
178 | 0 | unsigned int *sse) { |
179 | 0 | int sum; |
180 | 0 | uint16_t *src = CONVERT_TO_SHORTPTR(src8); |
181 | 0 | uint16_t *ref = CONVERT_TO_SHORTPTR(ref8); |
182 | 0 | highbd_10_variance_sse2(src, src_stride, ref, ref_stride, 16, 16, sse, &sum, |
183 | 0 | aom_highbd_calc16x16var_sse2, 16); |
184 | 0 | return *sse; |
185 | 0 | } |
186 | | |
187 | | unsigned int aom_highbd_12_mse16x16_sse2(const uint8_t *src8, int src_stride, |
188 | | const uint8_t *ref8, int ref_stride, |
189 | 0 | unsigned int *sse) { |
190 | 0 | int sum; |
191 | 0 | uint16_t *src = CONVERT_TO_SHORTPTR(src8); |
192 | 0 | uint16_t *ref = CONVERT_TO_SHORTPTR(ref8); |
193 | 0 | highbd_12_variance_sse2(src, src_stride, ref, ref_stride, 16, 16, sse, &sum, |
194 | 0 | aom_highbd_calc16x16var_sse2, 16); |
195 | 0 | return *sse; |
196 | 0 | } |
197 | | |
198 | | unsigned int aom_highbd_8_mse8x8_sse2(const uint8_t *src8, int src_stride, |
199 | | const uint8_t *ref8, int ref_stride, |
200 | 0 | unsigned int *sse) { |
201 | 0 | int sum; |
202 | 0 | uint16_t *src = CONVERT_TO_SHORTPTR(src8); |
203 | 0 | uint16_t *ref = CONVERT_TO_SHORTPTR(ref8); |
204 | 0 | highbd_8_variance_sse2(src, src_stride, ref, ref_stride, 8, 8, sse, &sum, |
205 | 0 | aom_highbd_calc8x8var_sse2, 8); |
206 | 0 | return *sse; |
207 | 0 | } |
208 | | |
209 | | unsigned int aom_highbd_10_mse8x8_sse2(const uint8_t *src8, int src_stride, |
210 | | const uint8_t *ref8, int ref_stride, |
211 | 0 | unsigned int *sse) { |
212 | 0 | int sum; |
213 | 0 | uint16_t *src = CONVERT_TO_SHORTPTR(src8); |
214 | 0 | uint16_t *ref = CONVERT_TO_SHORTPTR(ref8); |
215 | 0 | highbd_10_variance_sse2(src, src_stride, ref, ref_stride, 8, 8, sse, &sum, |
216 | 0 | aom_highbd_calc8x8var_sse2, 8); |
217 | 0 | return *sse; |
218 | 0 | } |
219 | | |
220 | | unsigned int aom_highbd_12_mse8x8_sse2(const uint8_t *src8, int src_stride, |
221 | | const uint8_t *ref8, int ref_stride, |
222 | 0 | unsigned int *sse) { |
223 | 0 | int sum; |
224 | 0 | uint16_t *src = CONVERT_TO_SHORTPTR(src8); |
225 | 0 | uint16_t *ref = CONVERT_TO_SHORTPTR(ref8); |
226 | 0 | highbd_12_variance_sse2(src, src_stride, ref, ref_stride, 8, 8, sse, &sum, |
227 | 0 | aom_highbd_calc8x8var_sse2, 8); |
228 | 0 | return *sse; |
229 | 0 | } |
230 | | |
231 | | // The 2 unused parameters are place holders for PIC enabled build. |
232 | | // These definitions are for functions defined in |
233 | | // highbd_subpel_variance_impl_sse2.asm |
234 | | #define DECL(w, opt) \ |
235 | | int aom_highbd_sub_pixel_variance##w##xh_##opt( \ |
236 | | const uint16_t *src, ptrdiff_t src_stride, int x_offset, int y_offset, \ |
237 | | const uint16_t *dst, ptrdiff_t dst_stride, int height, \ |
238 | | unsigned int *sse, void *unused0, void *unused); |
239 | | #define DECLS(opt) \ |
240 | | DECL(8, opt) \ |
241 | | DECL(16, opt) |
242 | | |
243 | | DECLS(sse2) |
244 | | |
245 | | #undef DECLS |
246 | | #undef DECL |
247 | | |
248 | | #define FN(w, h, wf, wlog2, hlog2, opt, cast) \ |
249 | | uint32_t aom_highbd_8_sub_pixel_variance##w##x##h##_##opt( \ |
250 | | const uint8_t *src8, int src_stride, int x_offset, int y_offset, \ |
251 | 0 | const uint8_t *dst8, int dst_stride, uint32_t *sse_ptr) { \ |
252 | 0 | uint16_t *src = CONVERT_TO_SHORTPTR(src8); \ |
253 | 0 | uint16_t *dst = CONVERT_TO_SHORTPTR(dst8); \ |
254 | 0 | int se = 0; \ |
255 | 0 | unsigned int sse = 0; \ |
256 | 0 | unsigned int sse2; \ |
257 | 0 | int row_rep = (w > 64) ? 2 : 1; \ |
258 | 0 | for (int wd_64 = 0; wd_64 < row_rep; wd_64++) { \ |
259 | 0 | src += wd_64 * 64; \ |
260 | 0 | dst += wd_64 * 64; \ |
261 | 0 | int se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt( \ |
262 | 0 | src, src_stride, x_offset, y_offset, dst, dst_stride, h, &sse2, \ |
263 | 0 | NULL, NULL); \ |
264 | 0 | se += se2; \ |
265 | 0 | sse += sse2; \ |
266 | 0 | if (w > wf) { \ |
267 | 0 | se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt( \ |
268 | 0 | src + wf, src_stride, x_offset, y_offset, dst + wf, dst_stride, h, \ |
269 | 0 | &sse2, NULL, NULL); \ |
270 | 0 | se += se2; \ |
271 | 0 | sse += sse2; \ |
272 | 0 | if (w > wf * 2) { \ |
273 | 0 | se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt( \ |
274 | 0 | src + 2 * wf, src_stride, x_offset, y_offset, dst + 2 * wf, \ |
275 | 0 | dst_stride, h, &sse2, NULL, NULL); \ |
276 | 0 | se += se2; \ |
277 | 0 | sse += sse2; \ |
278 | 0 | se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt( \ |
279 | 0 | src + 3 * wf, src_stride, x_offset, y_offset, dst + 3 * wf, \ |
280 | 0 | dst_stride, h, &sse2, NULL, NULL); \ |
281 | 0 | se += se2; \ |
282 | 0 | sse += sse2; \ |
283 | 0 | } \ |
284 | 0 | } \ |
285 | 0 | } \ |
286 | 0 | *sse_ptr = sse; \ |
287 | 0 | return sse - (uint32_t)((cast se * se) >> (wlog2 + hlog2)); \ |
288 | 0 | } \ |
289 | | \ |
290 | | uint32_t aom_highbd_10_sub_pixel_variance##w##x##h##_##opt( \ |
291 | | const uint8_t *src8, int src_stride, int x_offset, int y_offset, \ |
292 | 0 | const uint8_t *dst8, int dst_stride, uint32_t *sse_ptr) { \ |
293 | 0 | int64_t var; \ |
294 | 0 | uint32_t sse; \ |
295 | 0 | uint64_t long_sse = 0; \ |
296 | 0 | uint16_t *src = CONVERT_TO_SHORTPTR(src8); \ |
297 | 0 | uint16_t *dst = CONVERT_TO_SHORTPTR(dst8); \ |
298 | 0 | int se = 0; \ |
299 | 0 | int row_rep = (w > 64) ? 2 : 1; \ |
300 | 0 | for (int wd_64 = 0; wd_64 < row_rep; wd_64++) { \ |
301 | 0 | src += wd_64 * 64; \ |
302 | 0 | dst += wd_64 * 64; \ |
303 | 0 | int se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt( \ |
304 | 0 | src, src_stride, x_offset, y_offset, dst, dst_stride, h, &sse, NULL, \ |
305 | 0 | NULL); \ |
306 | 0 | se += se2; \ |
307 | 0 | long_sse += sse; \ |
308 | 0 | if (w > wf) { \ |
309 | 0 | uint32_t sse2; \ |
310 | 0 | se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt( \ |
311 | 0 | src + wf, src_stride, x_offset, y_offset, dst + wf, dst_stride, h, \ |
312 | 0 | &sse2, NULL, NULL); \ |
313 | 0 | se += se2; \ |
314 | 0 | long_sse += sse2; \ |
315 | 0 | if (w > wf * 2) { \ |
316 | 0 | se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt( \ |
317 | 0 | src + 2 * wf, src_stride, x_offset, y_offset, dst + 2 * wf, \ |
318 | 0 | dst_stride, h, &sse2, NULL, NULL); \ |
319 | 0 | se += se2; \ |
320 | 0 | long_sse += sse2; \ |
321 | 0 | se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt( \ |
322 | 0 | src + 3 * wf, src_stride, x_offset, y_offset, dst + 3 * wf, \ |
323 | 0 | dst_stride, h, &sse2, NULL, NULL); \ |
324 | 0 | se += se2; \ |
325 | 0 | long_sse += sse2; \ |
326 | 0 | } \ |
327 | 0 | } \ |
328 | 0 | } \ |
329 | 0 | se = ROUND_POWER_OF_TWO(se, 2); \ |
330 | 0 | sse = (uint32_t)ROUND_POWER_OF_TWO(long_sse, 4); \ |
331 | 0 | *sse_ptr = sse; \ |
332 | 0 | var = (int64_t)(sse) - ((cast se * se) >> (wlog2 + hlog2)); \ |
333 | 0 | return (var >= 0) ? (uint32_t)var : 0; \ |
334 | 0 | } \ |
335 | | \ |
336 | | uint32_t aom_highbd_12_sub_pixel_variance##w##x##h##_##opt( \ |
337 | | const uint8_t *src8, int src_stride, int x_offset, int y_offset, \ |
338 | 0 | const uint8_t *dst8, int dst_stride, uint32_t *sse_ptr) { \ |
339 | 0 | int start_row; \ |
340 | 0 | uint32_t sse; \ |
341 | 0 | int se = 0; \ |
342 | 0 | int64_t var; \ |
343 | 0 | uint64_t long_sse = 0; \ |
344 | 0 | uint16_t *src = CONVERT_TO_SHORTPTR(src8); \ |
345 | 0 | uint16_t *dst = CONVERT_TO_SHORTPTR(dst8); \ |
346 | 0 | int row_rep = (w > 64) ? 2 : 1; \ |
347 | 0 | for (start_row = 0; start_row < h; start_row += 16) { \ |
348 | 0 | uint32_t sse2; \ |
349 | 0 | int height = h - start_row < 16 ? h - start_row : 16; \ |
350 | 0 | uint16_t *src_tmp = src + (start_row * src_stride); \ |
351 | 0 | uint16_t *dst_tmp = dst + (start_row * dst_stride); \ |
352 | 0 | for (int wd_64 = 0; wd_64 < row_rep; wd_64++) { \ |
353 | 0 | src_tmp += wd_64 * 64; \ |
354 | 0 | dst_tmp += wd_64 * 64; \ |
355 | 0 | int se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt( \ |
356 | 0 | src_tmp, src_stride, x_offset, y_offset, dst_tmp, dst_stride, \ |
357 | 0 | height, &sse2, NULL, NULL); \ |
358 | 0 | se += se2; \ |
359 | 0 | long_sse += sse2; \ |
360 | 0 | if (w > wf) { \ |
361 | 0 | se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt( \ |
362 | 0 | src_tmp + wf, src_stride, x_offset, y_offset, dst_tmp + wf, \ |
363 | 0 | dst_stride, height, &sse2, NULL, NULL); \ |
364 | 0 | se += se2; \ |
365 | 0 | long_sse += sse2; \ |
366 | 0 | if (w > wf * 2) { \ |
367 | 0 | se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt( \ |
368 | 0 | src_tmp + 2 * wf, src_stride, x_offset, y_offset, \ |
369 | 0 | dst_tmp + 2 * wf, dst_stride, height, &sse2, NULL, NULL); \ |
370 | 0 | se += se2; \ |
371 | 0 | long_sse += sse2; \ |
372 | 0 | se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt( \ |
373 | 0 | src_tmp + 3 * wf, src_stride, x_offset, y_offset, \ |
374 | 0 | dst_tmp + 3 * wf, dst_stride, height, &sse2, NULL, NULL); \ |
375 | 0 | se += se2; \ |
376 | 0 | long_sse += sse2; \ |
377 | 0 | } \ |
378 | 0 | } \ |
379 | 0 | } \ |
380 | 0 | } \ |
381 | 0 | se = ROUND_POWER_OF_TWO(se, 4); \ |
382 | 0 | sse = (uint32_t)ROUND_POWER_OF_TWO(long_sse, 8); \ |
383 | 0 | *sse_ptr = sse; \ |
384 | 0 | var = (int64_t)(sse) - ((cast se * se) >> (wlog2 + hlog2)); \ |
385 | 0 | return (var >= 0) ? (uint32_t)var : 0; \ |
386 | 0 | } |
387 | | |
388 | | #if CONFIG_REALTIME_ONLY |
389 | | #define FNS(opt) \ |
390 | | FN(128, 128, 16, 7, 7, opt, (int64_t)) \ |
391 | | FN(128, 64, 16, 7, 6, opt, (int64_t)) \ |
392 | | FN(64, 128, 16, 6, 7, opt, (int64_t)) \ |
393 | | FN(64, 64, 16, 6, 6, opt, (int64_t)) \ |
394 | | FN(64, 32, 16, 6, 5, opt, (int64_t)) \ |
395 | | FN(32, 64, 16, 5, 6, opt, (int64_t)) \ |
396 | | FN(32, 32, 16, 5, 5, opt, (int64_t)) \ |
397 | | FN(32, 16, 16, 5, 4, opt, (int64_t)) \ |
398 | | FN(16, 32, 16, 4, 5, opt, (int64_t)) \ |
399 | | FN(16, 16, 16, 4, 4, opt, (int64_t)) \ |
400 | | FN(16, 8, 16, 4, 3, opt, (int64_t)) \ |
401 | | FN(8, 16, 8, 3, 4, opt, (int64_t)) \ |
402 | | FN(8, 8, 8, 3, 3, opt, (int64_t)) \ |
403 | | FN(8, 4, 8, 3, 2, opt, (int64_t)) |
404 | | #else // !CONFIG_REALTIME_ONLY |
405 | | #define FNS(opt) \ |
406 | | FN(128, 128, 16, 7, 7, opt, (int64_t)) \ |
407 | | FN(128, 64, 16, 7, 6, opt, (int64_t)) \ |
408 | | FN(64, 128, 16, 6, 7, opt, (int64_t)) \ |
409 | | FN(64, 64, 16, 6, 6, opt, (int64_t)) \ |
410 | | FN(64, 32, 16, 6, 5, opt, (int64_t)) \ |
411 | | FN(32, 64, 16, 5, 6, opt, (int64_t)) \ |
412 | | FN(32, 32, 16, 5, 5, opt, (int64_t)) \ |
413 | | FN(32, 16, 16, 5, 4, opt, (int64_t)) \ |
414 | | FN(16, 32, 16, 4, 5, opt, (int64_t)) \ |
415 | | FN(16, 16, 16, 4, 4, opt, (int64_t)) \ |
416 | | FN(16, 8, 16, 4, 3, opt, (int64_t)) \ |
417 | | FN(8, 16, 8, 3, 4, opt, (int64_t)) \ |
418 | | FN(8, 8, 8, 3, 3, opt, (int64_t)) \ |
419 | | FN(8, 4, 8, 3, 2, opt, (int64_t)) \ |
420 | | FN(16, 4, 16, 4, 2, opt, (int64_t)) \ |
421 | | FN(8, 32, 8, 3, 5, opt, (int64_t)) \ |
422 | | FN(32, 8, 16, 5, 3, opt, (int64_t)) \ |
423 | | FN(16, 64, 16, 4, 6, opt, (int64_t)) \ |
424 | | FN(64, 16, 16, 6, 4, opt, (int64_t)) |
425 | | #endif // CONFIG_REALTIME_ONLY |
426 | | |
427 | 0 | FNS(sse2) Unexecuted instantiation: aom_highbd_8_sub_pixel_variance128x128_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_variance128x128_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_variance128x128_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_variance128x64_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_variance128x64_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_variance128x64_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_variance64x128_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_variance64x128_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_variance64x128_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_variance64x64_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_variance64x64_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_variance64x64_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_variance64x32_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_variance64x32_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_variance64x32_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_variance32x64_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_variance32x64_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_variance32x64_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_variance32x32_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_variance32x32_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_variance32x32_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_variance32x16_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_variance32x16_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_variance32x16_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_variance16x32_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_variance16x32_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_variance16x32_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_variance16x16_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_variance16x16_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_variance16x16_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_variance16x8_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_variance16x8_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_variance16x8_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_variance8x16_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_variance8x16_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_variance8x16_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_variance8x8_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_variance8x8_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_variance8x8_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_variance8x4_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_variance8x4_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_variance8x4_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_variance16x4_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_variance16x4_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_variance16x4_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_variance8x32_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_variance8x32_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_variance8x32_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_variance32x8_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_variance32x8_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_variance32x8_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_variance16x64_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_variance16x64_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_variance16x64_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_variance64x16_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_variance64x16_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_variance64x16_sse2 |
428 | 0 |
|
429 | 0 | #undef FNS |
430 | 0 | #undef FN |
431 | 0 |
|
432 | 0 | // The 2 unused parameters are place holders for PIC enabled build. |
433 | 0 | #define DECL(w, opt) \ |
434 | 0 | int aom_highbd_sub_pixel_avg_variance##w##xh_##opt( \ |
435 | 0 | const uint16_t *src, ptrdiff_t src_stride, int x_offset, int y_offset, \ |
436 | 0 | const uint16_t *dst, ptrdiff_t dst_stride, const uint16_t *sec, \ |
437 | 0 | ptrdiff_t sec_stride, int height, unsigned int *sse, void *unused0, \ |
438 | 0 | void *unused); |
439 | 0 | #define DECLS(opt) \ |
440 | 0 | DECL(16, opt) \ |
441 | 0 | DECL(8, opt) |
442 | 0 |
|
443 | 0 | DECLS(sse2) |
444 | 0 | #undef DECL |
445 | 0 | #undef DECLS |
446 | 0 |
|
447 | 0 | #define FN(w, h, wf, wlog2, hlog2, opt, cast) \ |
448 | 0 | uint32_t aom_highbd_8_sub_pixel_avg_variance##w##x##h##_##opt( \ |
449 | 0 | const uint8_t *src8, int src_stride, int x_offset, int y_offset, \ |
450 | 0 | const uint8_t *dst8, int dst_stride, uint32_t *sse_ptr, \ |
451 | 0 | const uint8_t *sec8) { \ |
452 | 0 | uint32_t sse; \ |
453 | 0 | uint16_t *src = CONVERT_TO_SHORTPTR(src8); \ |
454 | 0 | uint16_t *dst = CONVERT_TO_SHORTPTR(dst8); \ |
455 | 0 | uint16_t *sec = CONVERT_TO_SHORTPTR(sec8); \ |
456 | 0 | int se = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt( \ |
457 | 0 | src, src_stride, x_offset, y_offset, dst, dst_stride, sec, w, h, &sse, \ |
458 | 0 | NULL, NULL); \ |
459 | 0 | if (w > wf) { \ |
460 | 0 | uint32_t sse2; \ |
461 | 0 | int se2 = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt( \ |
462 | 0 | src + wf, src_stride, x_offset, y_offset, dst + wf, dst_stride, \ |
463 | 0 | sec + wf, w, h, &sse2, NULL, NULL); \ |
464 | 0 | se += se2; \ |
465 | 0 | sse += sse2; \ |
466 | 0 | if (w > wf * 2) { \ |
467 | 0 | se2 = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt( \ |
468 | 0 | src + 2 * wf, src_stride, x_offset, y_offset, dst + 2 * wf, \ |
469 | 0 | dst_stride, sec + 2 * wf, w, h, &sse2, NULL, NULL); \ |
470 | 0 | se += se2; \ |
471 | 0 | sse += sse2; \ |
472 | 0 | se2 = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt( \ |
473 | 0 | src + 3 * wf, src_stride, x_offset, y_offset, dst + 3 * wf, \ |
474 | 0 | dst_stride, sec + 3 * wf, w, h, &sse2, NULL, NULL); \ |
475 | 0 | se += se2; \ |
476 | 0 | sse += sse2; \ |
477 | 0 | } \ |
478 | 0 | } \ |
479 | 0 | *sse_ptr = sse; \ |
480 | 0 | return sse - (uint32_t)((cast se * se) >> (wlog2 + hlog2)); \ |
481 | 0 | } \ |
482 | | \ |
483 | | uint32_t aom_highbd_10_sub_pixel_avg_variance##w##x##h##_##opt( \ |
484 | | const uint8_t *src8, int src_stride, int x_offset, int y_offset, \ |
485 | | const uint8_t *dst8, int dst_stride, uint32_t *sse_ptr, \ |
486 | 0 | const uint8_t *sec8) { \ |
487 | 0 | int64_t var; \ |
488 | 0 | uint32_t sse; \ |
489 | 0 | uint16_t *src = CONVERT_TO_SHORTPTR(src8); \ |
490 | 0 | uint16_t *dst = CONVERT_TO_SHORTPTR(dst8); \ |
491 | 0 | uint16_t *sec = CONVERT_TO_SHORTPTR(sec8); \ |
492 | 0 | int se = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt( \ |
493 | 0 | src, src_stride, x_offset, y_offset, dst, dst_stride, sec, w, h, &sse, \ |
494 | 0 | NULL, NULL); \ |
495 | 0 | if (w > wf) { \ |
496 | 0 | uint32_t sse2; \ |
497 | 0 | int se2 = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt( \ |
498 | 0 | src + wf, src_stride, x_offset, y_offset, dst + wf, dst_stride, \ |
499 | 0 | sec + wf, w, h, &sse2, NULL, NULL); \ |
500 | 0 | se += se2; \ |
501 | 0 | sse += sse2; \ |
502 | 0 | if (w > wf * 2) { \ |
503 | 0 | se2 = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt( \ |
504 | 0 | src + 2 * wf, src_stride, x_offset, y_offset, dst + 2 * wf, \ |
505 | 0 | dst_stride, sec + 2 * wf, w, h, &sse2, NULL, NULL); \ |
506 | 0 | se += se2; \ |
507 | 0 | sse += sse2; \ |
508 | 0 | se2 = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt( \ |
509 | 0 | src + 3 * wf, src_stride, x_offset, y_offset, dst + 3 * wf, \ |
510 | 0 | dst_stride, sec + 3 * wf, w, h, &sse2, NULL, NULL); \ |
511 | 0 | se += se2; \ |
512 | 0 | sse += sse2; \ |
513 | 0 | } \ |
514 | 0 | } \ |
515 | 0 | se = ROUND_POWER_OF_TWO(se, 2); \ |
516 | 0 | sse = ROUND_POWER_OF_TWO(sse, 4); \ |
517 | 0 | *sse_ptr = sse; \ |
518 | 0 | var = (int64_t)(sse) - ((cast se * se) >> (wlog2 + hlog2)); \ |
519 | 0 | return (var >= 0) ? (uint32_t)var : 0; \ |
520 | 0 | } \ |
521 | | \ |
522 | | uint32_t aom_highbd_12_sub_pixel_avg_variance##w##x##h##_##opt( \ |
523 | | const uint8_t *src8, int src_stride, int x_offset, int y_offset, \ |
524 | | const uint8_t *dst8, int dst_stride, uint32_t *sse_ptr, \ |
525 | 0 | const uint8_t *sec8) { \ |
526 | 0 | int start_row; \ |
527 | 0 | int64_t var; \ |
528 | 0 | uint32_t sse; \ |
529 | 0 | int se = 0; \ |
530 | 0 | uint64_t long_sse = 0; \ |
531 | 0 | uint16_t *src = CONVERT_TO_SHORTPTR(src8); \ |
532 | 0 | uint16_t *dst = CONVERT_TO_SHORTPTR(dst8); \ |
533 | 0 | uint16_t *sec = CONVERT_TO_SHORTPTR(sec8); \ |
534 | 0 | for (start_row = 0; start_row < h; start_row += 16) { \ |
535 | 0 | uint32_t sse2; \ |
536 | 0 | int height = h - start_row < 16 ? h - start_row : 16; \ |
537 | 0 | int se2 = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt( \ |
538 | 0 | src + (start_row * src_stride), src_stride, x_offset, y_offset, \ |
539 | 0 | dst + (start_row * dst_stride), dst_stride, sec + (start_row * w), \ |
540 | 0 | w, height, &sse2, NULL, NULL); \ |
541 | 0 | se += se2; \ |
542 | 0 | long_sse += sse2; \ |
543 | 0 | if (w > wf) { \ |
544 | 0 | se2 = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt( \ |
545 | 0 | src + wf + (start_row * src_stride), src_stride, x_offset, \ |
546 | 0 | y_offset, dst + wf + (start_row * dst_stride), dst_stride, \ |
547 | 0 | sec + wf + (start_row * w), w, height, &sse2, NULL, NULL); \ |
548 | 0 | se += se2; \ |
549 | 0 | long_sse += sse2; \ |
550 | 0 | if (w > wf * 2) { \ |
551 | 0 | se2 = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt( \ |
552 | 0 | src + 2 * wf + (start_row * src_stride), src_stride, x_offset, \ |
553 | 0 | y_offset, dst + 2 * wf + (start_row * dst_stride), dst_stride, \ |
554 | 0 | sec + 2 * wf + (start_row * w), w, height, &sse2, NULL, NULL); \ |
555 | 0 | se += se2; \ |
556 | 0 | long_sse += sse2; \ |
557 | 0 | se2 = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt( \ |
558 | 0 | src + 3 * wf + (start_row * src_stride), src_stride, x_offset, \ |
559 | 0 | y_offset, dst + 3 * wf + (start_row * dst_stride), dst_stride, \ |
560 | 0 | sec + 3 * wf + (start_row * w), w, height, &sse2, NULL, NULL); \ |
561 | 0 | se += se2; \ |
562 | 0 | long_sse += sse2; \ |
563 | 0 | } \ |
564 | 0 | } \ |
565 | 0 | } \ |
566 | 0 | se = ROUND_POWER_OF_TWO(se, 4); \ |
567 | 0 | sse = (uint32_t)ROUND_POWER_OF_TWO(long_sse, 8); \ |
568 | 0 | *sse_ptr = sse; \ |
569 | 0 | var = (int64_t)(sse) - ((cast se * se) >> (wlog2 + hlog2)); \ |
570 | 0 | return (var >= 0) ? (uint32_t)var : 0; \ |
571 | 0 | } |
572 | | |
573 | | #if CONFIG_REALTIME_ONLY |
574 | | #define FNS(opt) \ |
575 | | FN(64, 64, 16, 6, 6, opt, (int64_t)) \ |
576 | | FN(64, 32, 16, 6, 5, opt, (int64_t)) \ |
577 | | FN(32, 64, 16, 5, 6, opt, (int64_t)) \ |
578 | | FN(32, 32, 16, 5, 5, opt, (int64_t)) \ |
579 | | FN(32, 16, 16, 5, 4, opt, (int64_t)) \ |
580 | | FN(16, 32, 16, 4, 5, opt, (int64_t)) \ |
581 | | FN(16, 16, 16, 4, 4, opt, (int64_t)) \ |
582 | | FN(16, 8, 16, 4, 3, opt, (int64_t)) \ |
583 | | FN(8, 16, 8, 3, 4, opt, (int64_t)) \ |
584 | | FN(8, 8, 8, 3, 3, opt, (int64_t)) \ |
585 | | FN(8, 4, 8, 3, 2, opt, (int64_t)) |
586 | | #else // !CONFIG_REALTIME_ONLY |
587 | | #define FNS(opt) \ |
588 | | FN(64, 64, 16, 6, 6, opt, (int64_t)) \ |
589 | | FN(64, 32, 16, 6, 5, opt, (int64_t)) \ |
590 | | FN(32, 64, 16, 5, 6, opt, (int64_t)) \ |
591 | | FN(32, 32, 16, 5, 5, opt, (int64_t)) \ |
592 | | FN(32, 16, 16, 5, 4, opt, (int64_t)) \ |
593 | | FN(16, 32, 16, 4, 5, opt, (int64_t)) \ |
594 | | FN(16, 16, 16, 4, 4, opt, (int64_t)) \ |
595 | | FN(16, 8, 16, 4, 3, opt, (int64_t)) \ |
596 | | FN(8, 16, 8, 3, 4, opt, (int64_t)) \ |
597 | | FN(8, 8, 8, 3, 3, opt, (int64_t)) \ |
598 | | FN(8, 4, 8, 3, 2, opt, (int64_t)) \ |
599 | | FN(16, 4, 16, 4, 2, opt, (int64_t)) \ |
600 | | FN(8, 32, 8, 3, 5, opt, (int64_t)) \ |
601 | | FN(32, 8, 16, 5, 3, opt, (int64_t)) \ |
602 | | FN(16, 64, 16, 4, 6, opt, (int64_t)) \ |
603 | | FN(64, 16, 16, 6, 4, opt, (int64_t)) |
604 | | #endif // CONFIG_REALTIME_ONLY |
605 | | |
606 | 0 | FNS(sse2) Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance64x64_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance64x64_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance64x64_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance64x32_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance64x32_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance64x32_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance32x64_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance32x64_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance32x64_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance32x32_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance32x32_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance32x32_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance32x16_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance32x16_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance32x16_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance16x32_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance16x32_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance16x32_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance16x16_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance16x16_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance16x16_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance16x8_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance16x8_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance16x8_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance8x16_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance8x16_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance8x16_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance8x8_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance8x8_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance8x8_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance8x4_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance8x4_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance8x4_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance16x4_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance16x4_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance16x4_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance8x32_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance8x32_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance8x32_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance32x8_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance32x8_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance32x8_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance16x64_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance16x64_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance16x64_sse2 Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance64x16_sse2 Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance64x16_sse2 Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance64x16_sse2 |
607 | 0 |
|
608 | 0 | #undef FNS |
609 | 0 | #undef FN |
610 | 0 |
|
611 | 0 | static uint64_t mse_4xh_16bit_highbd_sse2(uint16_t *dst, int dstride, |
612 | 0 | uint16_t *src, int sstride, int h) { |
613 | 0 | uint64_t sum = 0; |
614 | 0 | __m128i reg0_4x16, reg1_4x16; |
615 | 0 | __m128i src_8x16; |
616 | 0 | __m128i dst_8x16; |
617 | 0 | __m128i res0_4x32, res1_4x32, res0_4x64, res1_4x64, res2_4x64, res3_4x64; |
618 | 0 | __m128i sub_result_8x16; |
619 | 0 | const __m128i zeros = _mm_setzero_si128(); |
620 | 0 | __m128i square_result = _mm_setzero_si128(); |
621 | 0 | for (int i = 0; i < h; i += 2) { |
622 | 0 | reg0_4x16 = _mm_loadl_epi64((__m128i const *)(&dst[(i + 0) * dstride])); |
623 | 0 | reg1_4x16 = _mm_loadl_epi64((__m128i const *)(&dst[(i + 1) * dstride])); |
624 | 0 | dst_8x16 = _mm_unpacklo_epi64(reg0_4x16, reg1_4x16); |
625 | |
|
626 | 0 | reg0_4x16 = _mm_loadl_epi64((__m128i const *)(&src[(i + 0) * sstride])); |
627 | 0 | reg1_4x16 = _mm_loadl_epi64((__m128i const *)(&src[(i + 1) * sstride])); |
628 | 0 | src_8x16 = _mm_unpacklo_epi64(reg0_4x16, reg1_4x16); |
629 | |
|
630 | 0 | sub_result_8x16 = _mm_sub_epi16(src_8x16, dst_8x16); |
631 | |
|
632 | 0 | res0_4x32 = _mm_unpacklo_epi16(sub_result_8x16, zeros); |
633 | 0 | res1_4x32 = _mm_unpackhi_epi16(sub_result_8x16, zeros); |
634 | |
|
635 | 0 | res0_4x32 = _mm_madd_epi16(res0_4x32, res0_4x32); |
636 | 0 | res1_4x32 = _mm_madd_epi16(res1_4x32, res1_4x32); |
637 | |
|
638 | 0 | res0_4x64 = _mm_unpacklo_epi32(res0_4x32, zeros); |
639 | 0 | res1_4x64 = _mm_unpackhi_epi32(res0_4x32, zeros); |
640 | 0 | res2_4x64 = _mm_unpacklo_epi32(res1_4x32, zeros); |
641 | 0 | res3_4x64 = _mm_unpackhi_epi32(res1_4x32, zeros); |
642 | |
|
643 | 0 | square_result = _mm_add_epi64( |
644 | 0 | square_result, |
645 | 0 | _mm_add_epi64( |
646 | 0 | _mm_add_epi64(_mm_add_epi64(res0_4x64, res1_4x64), res2_4x64), |
647 | 0 | res3_4x64)); |
648 | 0 | } |
649 | |
|
650 | 0 | const __m128i sum_1x64 = |
651 | 0 | _mm_add_epi64(square_result, _mm_srli_si128(square_result, 8)); |
652 | 0 | xx_storel_64(&sum, sum_1x64); |
653 | 0 | return sum; |
654 | 0 | } |
655 | | |
656 | | static uint64_t mse_8xh_16bit_highbd_sse2(uint16_t *dst, int dstride, |
657 | 0 | uint16_t *src, int sstride, int h) { |
658 | 0 | uint64_t sum = 0; |
659 | 0 | __m128i src_8x16; |
660 | 0 | __m128i dst_8x16; |
661 | 0 | __m128i res0_4x32, res1_4x32, res0_4x64, res1_4x64, res2_4x64, res3_4x64; |
662 | 0 | __m128i sub_result_8x16; |
663 | 0 | const __m128i zeros = _mm_setzero_si128(); |
664 | 0 | __m128i square_result = _mm_setzero_si128(); |
665 | |
|
666 | 0 | for (int i = 0; i < h; i++) { |
667 | 0 | dst_8x16 = _mm_loadu_si128((__m128i *)&dst[i * dstride]); |
668 | 0 | src_8x16 = _mm_loadu_si128((__m128i *)&src[i * sstride]); |
669 | |
|
670 | 0 | sub_result_8x16 = _mm_sub_epi16(src_8x16, dst_8x16); |
671 | |
|
672 | 0 | res0_4x32 = _mm_unpacklo_epi16(sub_result_8x16, zeros); |
673 | 0 | res1_4x32 = _mm_unpackhi_epi16(sub_result_8x16, zeros); |
674 | |
|
675 | 0 | res0_4x32 = _mm_madd_epi16(res0_4x32, res0_4x32); |
676 | 0 | res1_4x32 = _mm_madd_epi16(res1_4x32, res1_4x32); |
677 | |
|
678 | 0 | res0_4x64 = _mm_unpacklo_epi32(res0_4x32, zeros); |
679 | 0 | res1_4x64 = _mm_unpackhi_epi32(res0_4x32, zeros); |
680 | 0 | res2_4x64 = _mm_unpacklo_epi32(res1_4x32, zeros); |
681 | 0 | res3_4x64 = _mm_unpackhi_epi32(res1_4x32, zeros); |
682 | |
|
683 | 0 | square_result = _mm_add_epi64( |
684 | 0 | square_result, |
685 | 0 | _mm_add_epi64( |
686 | 0 | _mm_add_epi64(_mm_add_epi64(res0_4x64, res1_4x64), res2_4x64), |
687 | 0 | res3_4x64)); |
688 | 0 | } |
689 | |
|
690 | 0 | const __m128i sum_1x64 = |
691 | 0 | _mm_add_epi64(square_result, _mm_srli_si128(square_result, 8)); |
692 | 0 | xx_storel_64(&sum, sum_1x64); |
693 | 0 | return sum; |
694 | 0 | } |
695 | | |
696 | | uint64_t aom_mse_wxh_16bit_highbd_sse2(uint16_t *dst, int dstride, |
697 | | uint16_t *src, int sstride, int w, |
698 | 0 | int h) { |
699 | 0 | assert((w == 8 || w == 4) && (h == 8 || h == 4) && |
700 | 0 | "w=8/4 and h=8/4 must satisfy"); |
701 | 0 | switch (w) { |
702 | 0 | case 4: return mse_4xh_16bit_highbd_sse2(dst, dstride, src, sstride, h); |
703 | 0 | case 8: return mse_8xh_16bit_highbd_sse2(dst, dstride, src, sstride, h); |
704 | 0 | default: assert(0 && "unsupported width"); return -1; |
705 | 0 | } |
706 | 0 | } |