Coverage Report

Created: 2026-09-07 06:44

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/aom/aom_dsp/x86/highbd_variance_sse2.c
Line
Count
Source
1
/*
2
 * Copyright (c) 2016, Alliance for Open Media. All rights reserved.
3
 *
4
 * This source code is subject to the terms of the BSD 2 Clause License and
5
 * the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
6
 * was not distributed with this source code in the LICENSE file, you can
7
 * obtain it at www.aomedia.org/license/software. If the Alliance for Open
8
 * Media Patent License 1.0 was not distributed with this source code in the
9
 * PATENTS file, you can obtain it at www.aomedia.org/license/patent.
10
 */
11
12
#include <assert.h>
13
#include <emmintrin.h>  // SSE2
14
15
#include "config/aom_config.h"
16
#include "config/aom_dsp_rtcd.h"
17
18
#include "aom_dsp/x86/synonyms.h"
19
#include "aom_ports/mem.h"
20
21
#include "av1/common/filter.h"
22
#include "av1/common/reconinter.h"
23
24
typedef uint32_t (*high_variance_fn_t)(const uint16_t *src, int src_stride,
25
                                       const uint16_t *ref, int ref_stride,
26
                                       uint32_t *sse, int *sum);
27
28
uint32_t aom_highbd_calc8x8var_sse2(const uint16_t *src, int src_stride,
29
                                    const uint16_t *ref, int ref_stride,
30
                                    uint32_t *sse, int *sum);
31
32
uint32_t aom_highbd_calc16x16var_sse2(const uint16_t *src, int src_stride,
33
                                      const uint16_t *ref, int ref_stride,
34
                                      uint32_t *sse, int *sum);
35
36
static void highbd_8_variance_sse2(const uint16_t *src, int src_stride,
37
                                   const uint16_t *ref, int ref_stride, int w,
38
                                   int h, uint32_t *sse, int *sum,
39
0
                                   high_variance_fn_t var_fn, int block_size) {
40
0
  int i, j;
41
42
0
  *sse = 0;
43
0
  *sum = 0;
44
45
0
  for (i = 0; i < h; i += block_size) {
46
0
    for (j = 0; j < w; j += block_size) {
47
0
      unsigned int sse0;
48
0
      int sum0;
49
0
      var_fn(src + src_stride * i + j, src_stride, ref + ref_stride * i + j,
50
0
             ref_stride, &sse0, &sum0);
51
0
      *sse += sse0;
52
0
      *sum += sum0;
53
0
    }
54
0
  }
55
0
}
56
57
static void highbd_10_variance_sse2(const uint16_t *src, int src_stride,
58
                                    const uint16_t *ref, int ref_stride, int w,
59
                                    int h, uint32_t *sse, int *sum,
60
0
                                    high_variance_fn_t var_fn, int block_size) {
61
0
  int i, j;
62
0
  uint64_t sse_long = 0;
63
0
  int32_t sum_long = 0;
64
65
0
  for (i = 0; i < h; i += block_size) {
66
0
    for (j = 0; j < w; j += block_size) {
67
0
      unsigned int sse0;
68
0
      int sum0;
69
0
      var_fn(src + src_stride * i + j, src_stride, ref + ref_stride * i + j,
70
0
             ref_stride, &sse0, &sum0);
71
0
      sse_long += sse0;
72
0
      sum_long += sum0;
73
0
    }
74
0
  }
75
0
  *sum = ROUND_POWER_OF_TWO(sum_long, 2);
76
0
  *sse = (uint32_t)ROUND_POWER_OF_TWO(sse_long, 4);
77
0
}
78
79
static void highbd_12_variance_sse2(const uint16_t *src, int src_stride,
80
                                    const uint16_t *ref, int ref_stride, int w,
81
                                    int h, uint32_t *sse, int *sum,
82
0
                                    high_variance_fn_t var_fn, int block_size) {
83
0
  int i, j;
84
0
  uint64_t sse_long = 0;
85
0
  int32_t sum_long = 0;
86
87
0
  for (i = 0; i < h; i += block_size) {
88
0
    for (j = 0; j < w; j += block_size) {
89
0
      unsigned int sse0;
90
0
      int sum0;
91
0
      var_fn(src + src_stride * i + j, src_stride, ref + ref_stride * i + j,
92
0
             ref_stride, &sse0, &sum0);
93
0
      sse_long += sse0;
94
0
      sum_long += sum0;
95
0
    }
96
0
  }
97
0
  *sum = ROUND_POWER_OF_TWO(sum_long, 4);
98
0
  *sse = (uint32_t)ROUND_POWER_OF_TWO(sse_long, 8);
99
0
}
100
101
#define VAR_FN(w, h, block_size, shift)                                    \
102
  uint32_t aom_highbd_8_variance##w##x##h##_sse2(                          \
103
      const uint8_t *src8, int src_stride, const uint8_t *ref8,            \
104
0
      int ref_stride, uint32_t *sse) {                                     \
105
0
    int sum;                                                               \
106
0
    uint16_t *src = CONVERT_TO_SHORTPTR(src8);                             \
107
0
    uint16_t *ref = CONVERT_TO_SHORTPTR(ref8);                             \
108
0
    highbd_8_variance_sse2(                                                \
109
0
        src, src_stride, ref, ref_stride, w, h, sse, &sum,                 \
110
0
        aom_highbd_calc##block_size##x##block_size##var_sse2, block_size); \
111
0
    return *sse - (uint32_t)(((int64_t)sum * sum) >> shift);               \
112
0
  }                                                                        \
Unexecuted instantiation: aom_highbd_8_variance128x128_sse2
Unexecuted instantiation: aom_highbd_8_variance128x64_sse2
Unexecuted instantiation: aom_highbd_8_variance64x128_sse2
Unexecuted instantiation: aom_highbd_8_variance64x64_sse2
Unexecuted instantiation: aom_highbd_8_variance64x32_sse2
Unexecuted instantiation: aom_highbd_8_variance32x64_sse2
Unexecuted instantiation: aom_highbd_8_variance32x32_sse2
Unexecuted instantiation: aom_highbd_8_variance32x16_sse2
Unexecuted instantiation: aom_highbd_8_variance16x32_sse2
Unexecuted instantiation: aom_highbd_8_variance16x16_sse2
Unexecuted instantiation: aom_highbd_8_variance16x8_sse2
Unexecuted instantiation: aom_highbd_8_variance8x16_sse2
Unexecuted instantiation: aom_highbd_8_variance8x8_sse2
Unexecuted instantiation: aom_highbd_8_variance8x32_sse2
Unexecuted instantiation: aom_highbd_8_variance32x8_sse2
Unexecuted instantiation: aom_highbd_8_variance16x64_sse2
Unexecuted instantiation: aom_highbd_8_variance64x16_sse2
113
                                                                           \
114
  uint32_t aom_highbd_10_variance##w##x##h##_sse2(                         \
115
      const uint8_t *src8, int src_stride, const uint8_t *ref8,            \
116
0
      int ref_stride, uint32_t *sse) {                                     \
117
0
    int sum;                                                               \
118
0
    int64_t var;                                                           \
119
0
    uint16_t *src = CONVERT_TO_SHORTPTR(src8);                             \
120
0
    uint16_t *ref = CONVERT_TO_SHORTPTR(ref8);                             \
121
0
    highbd_10_variance_sse2(                                               \
122
0
        src, src_stride, ref, ref_stride, w, h, sse, &sum,                 \
123
0
        aom_highbd_calc##block_size##x##block_size##var_sse2, block_size); \
124
0
    var = (int64_t)(*sse) - (((int64_t)sum * sum) >> shift);               \
125
0
    return (var >= 0) ? (uint32_t)var : 0;                                 \
126
0
  }                                                                        \
Unexecuted instantiation: aom_highbd_10_variance128x128_sse2
Unexecuted instantiation: aom_highbd_10_variance128x64_sse2
Unexecuted instantiation: aom_highbd_10_variance64x128_sse2
Unexecuted instantiation: aom_highbd_10_variance64x64_sse2
Unexecuted instantiation: aom_highbd_10_variance64x32_sse2
Unexecuted instantiation: aom_highbd_10_variance32x64_sse2
Unexecuted instantiation: aom_highbd_10_variance32x32_sse2
Unexecuted instantiation: aom_highbd_10_variance32x16_sse2
Unexecuted instantiation: aom_highbd_10_variance16x32_sse2
Unexecuted instantiation: aom_highbd_10_variance16x16_sse2
Unexecuted instantiation: aom_highbd_10_variance16x8_sse2
Unexecuted instantiation: aom_highbd_10_variance8x16_sse2
Unexecuted instantiation: aom_highbd_10_variance8x8_sse2
Unexecuted instantiation: aom_highbd_10_variance8x32_sse2
Unexecuted instantiation: aom_highbd_10_variance32x8_sse2
Unexecuted instantiation: aom_highbd_10_variance16x64_sse2
Unexecuted instantiation: aom_highbd_10_variance64x16_sse2
127
                                                                           \
128
  uint32_t aom_highbd_12_variance##w##x##h##_sse2(                         \
129
      const uint8_t *src8, int src_stride, const uint8_t *ref8,            \
130
0
      int ref_stride, uint32_t *sse) {                                     \
131
0
    int sum;                                                               \
132
0
    int64_t var;                                                           \
133
0
    uint16_t *src = CONVERT_TO_SHORTPTR(src8);                             \
134
0
    uint16_t *ref = CONVERT_TO_SHORTPTR(ref8);                             \
135
0
    highbd_12_variance_sse2(                                               \
136
0
        src, src_stride, ref, ref_stride, w, h, sse, &sum,                 \
137
0
        aom_highbd_calc##block_size##x##block_size##var_sse2, block_size); \
138
0
    var = (int64_t)(*sse) - (((int64_t)sum * sum) >> shift);               \
139
0
    return (var >= 0) ? (uint32_t)var : 0;                                 \
140
0
  }
Unexecuted instantiation: aom_highbd_12_variance128x128_sse2
Unexecuted instantiation: aom_highbd_12_variance128x64_sse2
Unexecuted instantiation: aom_highbd_12_variance64x128_sse2
Unexecuted instantiation: aom_highbd_12_variance64x64_sse2
Unexecuted instantiation: aom_highbd_12_variance64x32_sse2
Unexecuted instantiation: aom_highbd_12_variance32x64_sse2
Unexecuted instantiation: aom_highbd_12_variance32x32_sse2
Unexecuted instantiation: aom_highbd_12_variance32x16_sse2
Unexecuted instantiation: aom_highbd_12_variance16x32_sse2
Unexecuted instantiation: aom_highbd_12_variance16x16_sse2
Unexecuted instantiation: aom_highbd_12_variance16x8_sse2
Unexecuted instantiation: aom_highbd_12_variance8x16_sse2
Unexecuted instantiation: aom_highbd_12_variance8x8_sse2
Unexecuted instantiation: aom_highbd_12_variance8x32_sse2
Unexecuted instantiation: aom_highbd_12_variance32x8_sse2
Unexecuted instantiation: aom_highbd_12_variance16x64_sse2
Unexecuted instantiation: aom_highbd_12_variance64x16_sse2
141
142
VAR_FN(128, 128, 16, 14)
143
VAR_FN(128, 64, 16, 13)
144
VAR_FN(64, 128, 16, 13)
145
VAR_FN(64, 64, 16, 12)
146
VAR_FN(64, 32, 16, 11)
147
VAR_FN(32, 64, 16, 11)
148
VAR_FN(32, 32, 16, 10)
149
VAR_FN(32, 16, 16, 9)
150
VAR_FN(16, 32, 16, 9)
151
VAR_FN(16, 16, 16, 8)
152
VAR_FN(16, 8, 8, 7)
153
VAR_FN(8, 16, 8, 7)
154
VAR_FN(8, 8, 8, 6)
155
156
#if !CONFIG_REALTIME_ONLY
157
VAR_FN(8, 32, 8, 8)
158
VAR_FN(32, 8, 8, 8)
159
VAR_FN(16, 64, 16, 10)
160
VAR_FN(64, 16, 16, 10)
161
#endif  // !CONFIG_REALTIME_ONLY
162
163
#undef VAR_FN
164
165
unsigned int aom_highbd_8_mse16x16_sse2(const uint8_t *src8, int src_stride,
166
                                        const uint8_t *ref8, int ref_stride,
167
0
                                        unsigned int *sse) {
168
0
  int sum;
169
0
  uint16_t *src = CONVERT_TO_SHORTPTR(src8);
170
0
  uint16_t *ref = CONVERT_TO_SHORTPTR(ref8);
171
0
  highbd_8_variance_sse2(src, src_stride, ref, ref_stride, 16, 16, sse, &sum,
172
0
                         aom_highbd_calc16x16var_sse2, 16);
173
0
  return *sse;
174
0
}
175
176
unsigned int aom_highbd_10_mse16x16_sse2(const uint8_t *src8, int src_stride,
177
                                         const uint8_t *ref8, int ref_stride,
178
0
                                         unsigned int *sse) {
179
0
  int sum;
180
0
  uint16_t *src = CONVERT_TO_SHORTPTR(src8);
181
0
  uint16_t *ref = CONVERT_TO_SHORTPTR(ref8);
182
0
  highbd_10_variance_sse2(src, src_stride, ref, ref_stride, 16, 16, sse, &sum,
183
0
                          aom_highbd_calc16x16var_sse2, 16);
184
0
  return *sse;
185
0
}
186
187
unsigned int aom_highbd_12_mse16x16_sse2(const uint8_t *src8, int src_stride,
188
                                         const uint8_t *ref8, int ref_stride,
189
0
                                         unsigned int *sse) {
190
0
  int sum;
191
0
  uint16_t *src = CONVERT_TO_SHORTPTR(src8);
192
0
  uint16_t *ref = CONVERT_TO_SHORTPTR(ref8);
193
0
  highbd_12_variance_sse2(src, src_stride, ref, ref_stride, 16, 16, sse, &sum,
194
0
                          aom_highbd_calc16x16var_sse2, 16);
195
0
  return *sse;
196
0
}
197
198
unsigned int aom_highbd_8_mse8x8_sse2(const uint8_t *src8, int src_stride,
199
                                      const uint8_t *ref8, int ref_stride,
200
0
                                      unsigned int *sse) {
201
0
  int sum;
202
0
  uint16_t *src = CONVERT_TO_SHORTPTR(src8);
203
0
  uint16_t *ref = CONVERT_TO_SHORTPTR(ref8);
204
0
  highbd_8_variance_sse2(src, src_stride, ref, ref_stride, 8, 8, sse, &sum,
205
0
                         aom_highbd_calc8x8var_sse2, 8);
206
0
  return *sse;
207
0
}
208
209
unsigned int aom_highbd_10_mse8x8_sse2(const uint8_t *src8, int src_stride,
210
                                       const uint8_t *ref8, int ref_stride,
211
0
                                       unsigned int *sse) {
212
0
  int sum;
213
0
  uint16_t *src = CONVERT_TO_SHORTPTR(src8);
214
0
  uint16_t *ref = CONVERT_TO_SHORTPTR(ref8);
215
0
  highbd_10_variance_sse2(src, src_stride, ref, ref_stride, 8, 8, sse, &sum,
216
0
                          aom_highbd_calc8x8var_sse2, 8);
217
0
  return *sse;
218
0
}
219
220
unsigned int aom_highbd_12_mse8x8_sse2(const uint8_t *src8, int src_stride,
221
                                       const uint8_t *ref8, int ref_stride,
222
0
                                       unsigned int *sse) {
223
0
  int sum;
224
0
  uint16_t *src = CONVERT_TO_SHORTPTR(src8);
225
0
  uint16_t *ref = CONVERT_TO_SHORTPTR(ref8);
226
0
  highbd_12_variance_sse2(src, src_stride, ref, ref_stride, 8, 8, sse, &sum,
227
0
                          aom_highbd_calc8x8var_sse2, 8);
228
0
  return *sse;
229
0
}
230
231
// The 2 unused parameters are place holders for PIC enabled build.
232
// These definitions are for functions defined in
233
// highbd_subpel_variance_impl_sse2.asm
234
#define DECL(w, opt)                                                         \
235
  int aom_highbd_sub_pixel_variance##w##xh_##opt(                            \
236
      const uint16_t *src, ptrdiff_t src_stride, int x_offset, int y_offset, \
237
      const uint16_t *dst, ptrdiff_t dst_stride, int height,                 \
238
      unsigned int *sse, void *unused0, void *unused);
239
#define DECLS(opt) \
240
  DECL(8, opt)     \
241
  DECL(16, opt)
242
243
DECLS(sse2)
244
245
#undef DECLS
246
#undef DECL
247
248
#define FN(w, h, wf, wlog2, hlog2, opt, cast)                                  \
249
  uint32_t aom_highbd_8_sub_pixel_variance##w##x##h##_##opt(                   \
250
      const uint8_t *src8, int src_stride, int x_offset, int y_offset,         \
251
0
      const uint8_t *dst8, int dst_stride, uint32_t *sse_ptr) {                \
252
0
    uint16_t *src = CONVERT_TO_SHORTPTR(src8);                                 \
253
0
    uint16_t *dst = CONVERT_TO_SHORTPTR(dst8);                                 \
254
0
    int se = 0;                                                                \
255
0
    unsigned int sse = 0;                                                      \
256
0
    unsigned int sse2;                                                         \
257
0
    int row_rep = (w > 64) ? 2 : 1;                                            \
258
0
    for (int wd_64 = 0; wd_64 < row_rep; wd_64++) {                            \
259
0
      src += wd_64 * 64;                                                       \
260
0
      dst += wd_64 * 64;                                                       \
261
0
      int se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt(                   \
262
0
          src, src_stride, x_offset, y_offset, dst, dst_stride, h, &sse2,      \
263
0
          NULL, NULL);                                                         \
264
0
      se += se2;                                                               \
265
0
      sse += sse2;                                                             \
266
0
      if (w > wf) {                                                            \
267
0
        se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt(                     \
268
0
            src + wf, src_stride, x_offset, y_offset, dst + wf, dst_stride, h, \
269
0
            &sse2, NULL, NULL);                                                \
270
0
        se += se2;                                                             \
271
0
        sse += sse2;                                                           \
272
0
        if (w > wf * 2) {                                                      \
273
0
          se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt(                   \
274
0
              src + 2 * wf, src_stride, x_offset, y_offset, dst + 2 * wf,      \
275
0
              dst_stride, h, &sse2, NULL, NULL);                               \
276
0
          se += se2;                                                           \
277
0
          sse += sse2;                                                         \
278
0
          se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt(                   \
279
0
              src + 3 * wf, src_stride, x_offset, y_offset, dst + 3 * wf,      \
280
0
              dst_stride, h, &sse2, NULL, NULL);                               \
281
0
          se += se2;                                                           \
282
0
          sse += sse2;                                                         \
283
0
        }                                                                      \
284
0
      }                                                                        \
285
0
    }                                                                          \
286
0
    *sse_ptr = sse;                                                            \
287
0
    return sse - (uint32_t)((cast se * se) >> (wlog2 + hlog2));                \
288
0
  }                                                                            \
289
                                                                               \
290
  uint32_t aom_highbd_10_sub_pixel_variance##w##x##h##_##opt(                  \
291
      const uint8_t *src8, int src_stride, int x_offset, int y_offset,         \
292
0
      const uint8_t *dst8, int dst_stride, uint32_t *sse_ptr) {                \
293
0
    int64_t var;                                                               \
294
0
    uint32_t sse;                                                              \
295
0
    uint64_t long_sse = 0;                                                     \
296
0
    uint16_t *src = CONVERT_TO_SHORTPTR(src8);                                 \
297
0
    uint16_t *dst = CONVERT_TO_SHORTPTR(dst8);                                 \
298
0
    int se = 0;                                                                \
299
0
    int row_rep = (w > 64) ? 2 : 1;                                            \
300
0
    for (int wd_64 = 0; wd_64 < row_rep; wd_64++) {                            \
301
0
      src += wd_64 * 64;                                                       \
302
0
      dst += wd_64 * 64;                                                       \
303
0
      int se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt(                   \
304
0
          src, src_stride, x_offset, y_offset, dst, dst_stride, h, &sse, NULL, \
305
0
          NULL);                                                               \
306
0
      se += se2;                                                               \
307
0
      long_sse += sse;                                                         \
308
0
      if (w > wf) {                                                            \
309
0
        uint32_t sse2;                                                         \
310
0
        se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt(                     \
311
0
            src + wf, src_stride, x_offset, y_offset, dst + wf, dst_stride, h, \
312
0
            &sse2, NULL, NULL);                                                \
313
0
        se += se2;                                                             \
314
0
        long_sse += sse2;                                                      \
315
0
        if (w > wf * 2) {                                                      \
316
0
          se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt(                   \
317
0
              src + 2 * wf, src_stride, x_offset, y_offset, dst + 2 * wf,      \
318
0
              dst_stride, h, &sse2, NULL, NULL);                               \
319
0
          se += se2;                                                           \
320
0
          long_sse += sse2;                                                    \
321
0
          se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt(                   \
322
0
              src + 3 * wf, src_stride, x_offset, y_offset, dst + 3 * wf,      \
323
0
              dst_stride, h, &sse2, NULL, NULL);                               \
324
0
          se += se2;                                                           \
325
0
          long_sse += sse2;                                                    \
326
0
        }                                                                      \
327
0
      }                                                                        \
328
0
    }                                                                          \
329
0
    se = ROUND_POWER_OF_TWO(se, 2);                                            \
330
0
    sse = (uint32_t)ROUND_POWER_OF_TWO(long_sse, 4);                           \
331
0
    *sse_ptr = sse;                                                            \
332
0
    var = (int64_t)(sse) - ((cast se * se) >> (wlog2 + hlog2));                \
333
0
    return (var >= 0) ? (uint32_t)var : 0;                                     \
334
0
  }                                                                            \
335
                                                                               \
336
  uint32_t aom_highbd_12_sub_pixel_variance##w##x##h##_##opt(                  \
337
      const uint8_t *src8, int src_stride, int x_offset, int y_offset,         \
338
0
      const uint8_t *dst8, int dst_stride, uint32_t *sse_ptr) {                \
339
0
    int start_row;                                                             \
340
0
    uint32_t sse;                                                              \
341
0
    int se = 0;                                                                \
342
0
    int64_t var;                                                               \
343
0
    uint64_t long_sse = 0;                                                     \
344
0
    uint16_t *src = CONVERT_TO_SHORTPTR(src8);                                 \
345
0
    uint16_t *dst = CONVERT_TO_SHORTPTR(dst8);                                 \
346
0
    int row_rep = (w > 64) ? 2 : 1;                                            \
347
0
    for (start_row = 0; start_row < h; start_row += 16) {                      \
348
0
      uint32_t sse2;                                                           \
349
0
      int height = h - start_row < 16 ? h - start_row : 16;                    \
350
0
      uint16_t *src_tmp = src + (start_row * src_stride);                      \
351
0
      uint16_t *dst_tmp = dst + (start_row * dst_stride);                      \
352
0
      for (int wd_64 = 0; wd_64 < row_rep; wd_64++) {                          \
353
0
        src_tmp += wd_64 * 64;                                                 \
354
0
        dst_tmp += wd_64 * 64;                                                 \
355
0
        int se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt(                 \
356
0
            src_tmp, src_stride, x_offset, y_offset, dst_tmp, dst_stride,      \
357
0
            height, &sse2, NULL, NULL);                                        \
358
0
        se += se2;                                                             \
359
0
        long_sse += sse2;                                                      \
360
0
        if (w > wf) {                                                          \
361
0
          se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt(                   \
362
0
              src_tmp + wf, src_stride, x_offset, y_offset, dst_tmp + wf,      \
363
0
              dst_stride, height, &sse2, NULL, NULL);                          \
364
0
          se += se2;                                                           \
365
0
          long_sse += sse2;                                                    \
366
0
          if (w > wf * 2) {                                                    \
367
0
            se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt(                 \
368
0
                src_tmp + 2 * wf, src_stride, x_offset, y_offset,              \
369
0
                dst_tmp + 2 * wf, dst_stride, height, &sse2, NULL, NULL);      \
370
0
            se += se2;                                                         \
371
0
            long_sse += sse2;                                                  \
372
0
            se2 = aom_highbd_sub_pixel_variance##wf##xh_##opt(                 \
373
0
                src_tmp + 3 * wf, src_stride, x_offset, y_offset,              \
374
0
                dst_tmp + 3 * wf, dst_stride, height, &sse2, NULL, NULL);      \
375
0
            se += se2;                                                         \
376
0
            long_sse += sse2;                                                  \
377
0
          }                                                                    \
378
0
        }                                                                      \
379
0
      }                                                                        \
380
0
    }                                                                          \
381
0
    se = ROUND_POWER_OF_TWO(se, 4);                                            \
382
0
    sse = (uint32_t)ROUND_POWER_OF_TWO(long_sse, 8);                           \
383
0
    *sse_ptr = sse;                                                            \
384
0
    var = (int64_t)(sse) - ((cast se * se) >> (wlog2 + hlog2));                \
385
0
    return (var >= 0) ? (uint32_t)var : 0;                                     \
386
0
  }
387
388
#if CONFIG_REALTIME_ONLY
389
#define FNS(opt)                         \
390
  FN(128, 128, 16, 7, 7, opt, (int64_t)) \
391
  FN(128, 64, 16, 7, 6, opt, (int64_t))  \
392
  FN(64, 128, 16, 6, 7, opt, (int64_t))  \
393
  FN(64, 64, 16, 6, 6, opt, (int64_t))   \
394
  FN(64, 32, 16, 6, 5, opt, (int64_t))   \
395
  FN(32, 64, 16, 5, 6, opt, (int64_t))   \
396
  FN(32, 32, 16, 5, 5, opt, (int64_t))   \
397
  FN(32, 16, 16, 5, 4, opt, (int64_t))   \
398
  FN(16, 32, 16, 4, 5, opt, (int64_t))   \
399
  FN(16, 16, 16, 4, 4, opt, (int64_t))   \
400
  FN(16, 8, 16, 4, 3, opt, (int64_t))    \
401
  FN(8, 16, 8, 3, 4, opt, (int64_t))     \
402
  FN(8, 8, 8, 3, 3, opt, (int64_t))      \
403
  FN(8, 4, 8, 3, 2, opt, (int64_t))
404
#else  // !CONFIG_REALTIME_ONLY
405
#define FNS(opt)                         \
406
  FN(128, 128, 16, 7, 7, opt, (int64_t)) \
407
  FN(128, 64, 16, 7, 6, opt, (int64_t))  \
408
  FN(64, 128, 16, 6, 7, opt, (int64_t))  \
409
  FN(64, 64, 16, 6, 6, opt, (int64_t))   \
410
  FN(64, 32, 16, 6, 5, opt, (int64_t))   \
411
  FN(32, 64, 16, 5, 6, opt, (int64_t))   \
412
  FN(32, 32, 16, 5, 5, opt, (int64_t))   \
413
  FN(32, 16, 16, 5, 4, opt, (int64_t))   \
414
  FN(16, 32, 16, 4, 5, opt, (int64_t))   \
415
  FN(16, 16, 16, 4, 4, opt, (int64_t))   \
416
  FN(16, 8, 16, 4, 3, opt, (int64_t))    \
417
  FN(8, 16, 8, 3, 4, opt, (int64_t))     \
418
  FN(8, 8, 8, 3, 3, opt, (int64_t))      \
419
  FN(8, 4, 8, 3, 2, opt, (int64_t))      \
420
  FN(16, 4, 16, 4, 2, opt, (int64_t))    \
421
  FN(8, 32, 8, 3, 5, opt, (int64_t))     \
422
  FN(32, 8, 16, 5, 3, opt, (int64_t))    \
423
  FN(16, 64, 16, 4, 6, opt, (int64_t))   \
424
  FN(64, 16, 16, 6, 4, opt, (int64_t))
425
#endif  // CONFIG_REALTIME_ONLY
426
427
0
FNS(sse2)
Unexecuted instantiation: aom_highbd_8_sub_pixel_variance128x128_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_variance128x128_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_variance128x128_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_variance128x64_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_variance128x64_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_variance128x64_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_variance64x128_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_variance64x128_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_variance64x128_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_variance64x64_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_variance64x64_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_variance64x64_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_variance64x32_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_variance64x32_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_variance64x32_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_variance32x64_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_variance32x64_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_variance32x64_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_variance32x32_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_variance32x32_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_variance32x32_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_variance32x16_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_variance32x16_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_variance32x16_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_variance16x32_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_variance16x32_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_variance16x32_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_variance16x16_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_variance16x16_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_variance16x16_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_variance16x8_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_variance16x8_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_variance16x8_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_variance8x16_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_variance8x16_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_variance8x16_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_variance8x8_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_variance8x8_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_variance8x8_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_variance8x4_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_variance8x4_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_variance8x4_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_variance16x4_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_variance16x4_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_variance16x4_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_variance8x32_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_variance8x32_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_variance8x32_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_variance32x8_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_variance32x8_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_variance32x8_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_variance16x64_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_variance16x64_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_variance16x64_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_variance64x16_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_variance64x16_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_variance64x16_sse2
428
0
429
0
#undef FNS
430
0
#undef FN
431
0
432
0
// The 2 unused parameters are place holders for PIC enabled build.
433
0
#define DECL(w, opt)                                                         \
434
0
  int aom_highbd_sub_pixel_avg_variance##w##xh_##opt(                        \
435
0
      const uint16_t *src, ptrdiff_t src_stride, int x_offset, int y_offset, \
436
0
      const uint16_t *dst, ptrdiff_t dst_stride, const uint16_t *sec,        \
437
0
      ptrdiff_t sec_stride, int height, unsigned int *sse, void *unused0,    \
438
0
      void *unused);
439
0
#define DECLS(opt) \
440
0
  DECL(16, opt)    \
441
0
  DECL(8, opt)
442
0
443
0
DECLS(sse2)
444
0
#undef DECL
445
0
#undef DECLS
446
0
447
0
#define FN(w, h, wf, wlog2, hlog2, opt, cast)                                  \
448
0
  uint32_t aom_highbd_8_sub_pixel_avg_variance##w##x##h##_##opt(               \
449
0
      const uint8_t *src8, int src_stride, int x_offset, int y_offset,         \
450
0
      const uint8_t *dst8, int dst_stride, uint32_t *sse_ptr,                  \
451
0
      const uint8_t *sec8) {                                                   \
452
0
    uint32_t sse;                                                              \
453
0
    uint16_t *src = CONVERT_TO_SHORTPTR(src8);                                 \
454
0
    uint16_t *dst = CONVERT_TO_SHORTPTR(dst8);                                 \
455
0
    uint16_t *sec = CONVERT_TO_SHORTPTR(sec8);                                 \
456
0
    int se = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt(                  \
457
0
        src, src_stride, x_offset, y_offset, dst, dst_stride, sec, w, h, &sse, \
458
0
        NULL, NULL);                                                           \
459
0
    if (w > wf) {                                                              \
460
0
      uint32_t sse2;                                                           \
461
0
      int se2 = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt(               \
462
0
          src + wf, src_stride, x_offset, y_offset, dst + wf, dst_stride,      \
463
0
          sec + wf, w, h, &sse2, NULL, NULL);                                  \
464
0
      se += se2;                                                               \
465
0
      sse += sse2;                                                             \
466
0
      if (w > wf * 2) {                                                        \
467
0
        se2 = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt(                 \
468
0
            src + 2 * wf, src_stride, x_offset, y_offset, dst + 2 * wf,        \
469
0
            dst_stride, sec + 2 * wf, w, h, &sse2, NULL, NULL);                \
470
0
        se += se2;                                                             \
471
0
        sse += sse2;                                                           \
472
0
        se2 = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt(                 \
473
0
            src + 3 * wf, src_stride, x_offset, y_offset, dst + 3 * wf,        \
474
0
            dst_stride, sec + 3 * wf, w, h, &sse2, NULL, NULL);                \
475
0
        se += se2;                                                             \
476
0
        sse += sse2;                                                           \
477
0
      }                                                                        \
478
0
    }                                                                          \
479
0
    *sse_ptr = sse;                                                            \
480
0
    return sse - (uint32_t)((cast se * se) >> (wlog2 + hlog2));                \
481
0
  }                                                                            \
482
                                                                               \
483
  uint32_t aom_highbd_10_sub_pixel_avg_variance##w##x##h##_##opt(              \
484
      const uint8_t *src8, int src_stride, int x_offset, int y_offset,         \
485
      const uint8_t *dst8, int dst_stride, uint32_t *sse_ptr,                  \
486
0
      const uint8_t *sec8) {                                                   \
487
0
    int64_t var;                                                               \
488
0
    uint32_t sse;                                                              \
489
0
    uint16_t *src = CONVERT_TO_SHORTPTR(src8);                                 \
490
0
    uint16_t *dst = CONVERT_TO_SHORTPTR(dst8);                                 \
491
0
    uint16_t *sec = CONVERT_TO_SHORTPTR(sec8);                                 \
492
0
    int se = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt(                  \
493
0
        src, src_stride, x_offset, y_offset, dst, dst_stride, sec, w, h, &sse, \
494
0
        NULL, NULL);                                                           \
495
0
    if (w > wf) {                                                              \
496
0
      uint32_t sse2;                                                           \
497
0
      int se2 = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt(               \
498
0
          src + wf, src_stride, x_offset, y_offset, dst + wf, dst_stride,      \
499
0
          sec + wf, w, h, &sse2, NULL, NULL);                                  \
500
0
      se += se2;                                                               \
501
0
      sse += sse2;                                                             \
502
0
      if (w > wf * 2) {                                                        \
503
0
        se2 = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt(                 \
504
0
            src + 2 * wf, src_stride, x_offset, y_offset, dst + 2 * wf,        \
505
0
            dst_stride, sec + 2 * wf, w, h, &sse2, NULL, NULL);                \
506
0
        se += se2;                                                             \
507
0
        sse += sse2;                                                           \
508
0
        se2 = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt(                 \
509
0
            src + 3 * wf, src_stride, x_offset, y_offset, dst + 3 * wf,        \
510
0
            dst_stride, sec + 3 * wf, w, h, &sse2, NULL, NULL);                \
511
0
        se += se2;                                                             \
512
0
        sse += sse2;                                                           \
513
0
      }                                                                        \
514
0
    }                                                                          \
515
0
    se = ROUND_POWER_OF_TWO(se, 2);                                            \
516
0
    sse = ROUND_POWER_OF_TWO(sse, 4);                                          \
517
0
    *sse_ptr = sse;                                                            \
518
0
    var = (int64_t)(sse) - ((cast se * se) >> (wlog2 + hlog2));                \
519
0
    return (var >= 0) ? (uint32_t)var : 0;                                     \
520
0
  }                                                                            \
521
                                                                               \
522
  uint32_t aom_highbd_12_sub_pixel_avg_variance##w##x##h##_##opt(              \
523
      const uint8_t *src8, int src_stride, int x_offset, int y_offset,         \
524
      const uint8_t *dst8, int dst_stride, uint32_t *sse_ptr,                  \
525
0
      const uint8_t *sec8) {                                                   \
526
0
    int start_row;                                                             \
527
0
    int64_t var;                                                               \
528
0
    uint32_t sse;                                                              \
529
0
    int se = 0;                                                                \
530
0
    uint64_t long_sse = 0;                                                     \
531
0
    uint16_t *src = CONVERT_TO_SHORTPTR(src8);                                 \
532
0
    uint16_t *dst = CONVERT_TO_SHORTPTR(dst8);                                 \
533
0
    uint16_t *sec = CONVERT_TO_SHORTPTR(sec8);                                 \
534
0
    for (start_row = 0; start_row < h; start_row += 16) {                      \
535
0
      uint32_t sse2;                                                           \
536
0
      int height = h - start_row < 16 ? h - start_row : 16;                    \
537
0
      int se2 = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt(               \
538
0
          src + (start_row * src_stride), src_stride, x_offset, y_offset,      \
539
0
          dst + (start_row * dst_stride), dst_stride, sec + (start_row * w),   \
540
0
          w, height, &sse2, NULL, NULL);                                       \
541
0
      se += se2;                                                               \
542
0
      long_sse += sse2;                                                        \
543
0
      if (w > wf) {                                                            \
544
0
        se2 = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt(                 \
545
0
            src + wf + (start_row * src_stride), src_stride, x_offset,         \
546
0
            y_offset, dst + wf + (start_row * dst_stride), dst_stride,         \
547
0
            sec + wf + (start_row * w), w, height, &sse2, NULL, NULL);         \
548
0
        se += se2;                                                             \
549
0
        long_sse += sse2;                                                      \
550
0
        if (w > wf * 2) {                                                      \
551
0
          se2 = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt(               \
552
0
              src + 2 * wf + (start_row * src_stride), src_stride, x_offset,   \
553
0
              y_offset, dst + 2 * wf + (start_row * dst_stride), dst_stride,   \
554
0
              sec + 2 * wf + (start_row * w), w, height, &sse2, NULL, NULL);   \
555
0
          se += se2;                                                           \
556
0
          long_sse += sse2;                                                    \
557
0
          se2 = aom_highbd_sub_pixel_avg_variance##wf##xh_##opt(               \
558
0
              src + 3 * wf + (start_row * src_stride), src_stride, x_offset,   \
559
0
              y_offset, dst + 3 * wf + (start_row * dst_stride), dst_stride,   \
560
0
              sec + 3 * wf + (start_row * w), w, height, &sse2, NULL, NULL);   \
561
0
          se += se2;                                                           \
562
0
          long_sse += sse2;                                                    \
563
0
        }                                                                      \
564
0
      }                                                                        \
565
0
    }                                                                          \
566
0
    se = ROUND_POWER_OF_TWO(se, 4);                                            \
567
0
    sse = (uint32_t)ROUND_POWER_OF_TWO(long_sse, 8);                           \
568
0
    *sse_ptr = sse;                                                            \
569
0
    var = (int64_t)(sse) - ((cast se * se) >> (wlog2 + hlog2));                \
570
0
    return (var >= 0) ? (uint32_t)var : 0;                                     \
571
0
  }
572
573
#if CONFIG_REALTIME_ONLY
574
#define FNS(opt)                       \
575
  FN(64, 64, 16, 6, 6, opt, (int64_t)) \
576
  FN(64, 32, 16, 6, 5, opt, (int64_t)) \
577
  FN(32, 64, 16, 5, 6, opt, (int64_t)) \
578
  FN(32, 32, 16, 5, 5, opt, (int64_t)) \
579
  FN(32, 16, 16, 5, 4, opt, (int64_t)) \
580
  FN(16, 32, 16, 4, 5, opt, (int64_t)) \
581
  FN(16, 16, 16, 4, 4, opt, (int64_t)) \
582
  FN(16, 8, 16, 4, 3, opt, (int64_t))  \
583
  FN(8, 16, 8, 3, 4, opt, (int64_t))   \
584
  FN(8, 8, 8, 3, 3, opt, (int64_t))    \
585
  FN(8, 4, 8, 3, 2, opt, (int64_t))
586
#else  // !CONFIG_REALTIME_ONLY
587
#define FNS(opt)                       \
588
  FN(64, 64, 16, 6, 6, opt, (int64_t)) \
589
  FN(64, 32, 16, 6, 5, opt, (int64_t)) \
590
  FN(32, 64, 16, 5, 6, opt, (int64_t)) \
591
  FN(32, 32, 16, 5, 5, opt, (int64_t)) \
592
  FN(32, 16, 16, 5, 4, opt, (int64_t)) \
593
  FN(16, 32, 16, 4, 5, opt, (int64_t)) \
594
  FN(16, 16, 16, 4, 4, opt, (int64_t)) \
595
  FN(16, 8, 16, 4, 3, opt, (int64_t))  \
596
  FN(8, 16, 8, 3, 4, opt, (int64_t))   \
597
  FN(8, 8, 8, 3, 3, opt, (int64_t))    \
598
  FN(8, 4, 8, 3, 2, opt, (int64_t))    \
599
  FN(16, 4, 16, 4, 2, opt, (int64_t))  \
600
  FN(8, 32, 8, 3, 5, opt, (int64_t))   \
601
  FN(32, 8, 16, 5, 3, opt, (int64_t))  \
602
  FN(16, 64, 16, 4, 6, opt, (int64_t)) \
603
  FN(64, 16, 16, 6, 4, opt, (int64_t))
604
#endif  // CONFIG_REALTIME_ONLY
605
606
0
FNS(sse2)
Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance64x64_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance64x64_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance64x64_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance64x32_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance64x32_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance64x32_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance32x64_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance32x64_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance32x64_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance32x32_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance32x32_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance32x32_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance32x16_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance32x16_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance32x16_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance16x32_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance16x32_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance16x32_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance16x16_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance16x16_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance16x16_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance16x8_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance16x8_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance16x8_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance8x16_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance8x16_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance8x16_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance8x8_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance8x8_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance8x8_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance8x4_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance8x4_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance8x4_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance16x4_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance16x4_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance16x4_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance8x32_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance8x32_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance8x32_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance32x8_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance32x8_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance32x8_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance16x64_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance16x64_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance16x64_sse2
Unexecuted instantiation: aom_highbd_8_sub_pixel_avg_variance64x16_sse2
Unexecuted instantiation: aom_highbd_10_sub_pixel_avg_variance64x16_sse2
Unexecuted instantiation: aom_highbd_12_sub_pixel_avg_variance64x16_sse2
607
0
608
0
#undef FNS
609
0
#undef FN
610
0
611
0
static uint64_t mse_4xh_16bit_highbd_sse2(uint16_t *dst, int dstride,
612
0
                                          uint16_t *src, int sstride, int h) {
613
0
  uint64_t sum = 0;
614
0
  __m128i reg0_4x16, reg1_4x16;
615
0
  __m128i src_8x16;
616
0
  __m128i dst_8x16;
617
0
  __m128i res0_4x32, res1_4x32, res0_4x64, res1_4x64, res2_4x64, res3_4x64;
618
0
  __m128i sub_result_8x16;
619
0
  const __m128i zeros = _mm_setzero_si128();
620
0
  __m128i square_result = _mm_setzero_si128();
621
0
  for (int i = 0; i < h; i += 2) {
622
0
    reg0_4x16 = _mm_loadl_epi64((__m128i const *)(&dst[(i + 0) * dstride]));
623
0
    reg1_4x16 = _mm_loadl_epi64((__m128i const *)(&dst[(i + 1) * dstride]));
624
0
    dst_8x16 = _mm_unpacklo_epi64(reg0_4x16, reg1_4x16);
625
626
0
    reg0_4x16 = _mm_loadl_epi64((__m128i const *)(&src[(i + 0) * sstride]));
627
0
    reg1_4x16 = _mm_loadl_epi64((__m128i const *)(&src[(i + 1) * sstride]));
628
0
    src_8x16 = _mm_unpacklo_epi64(reg0_4x16, reg1_4x16);
629
630
0
    sub_result_8x16 = _mm_sub_epi16(src_8x16, dst_8x16);
631
632
0
    res0_4x32 = _mm_unpacklo_epi16(sub_result_8x16, zeros);
633
0
    res1_4x32 = _mm_unpackhi_epi16(sub_result_8x16, zeros);
634
635
0
    res0_4x32 = _mm_madd_epi16(res0_4x32, res0_4x32);
636
0
    res1_4x32 = _mm_madd_epi16(res1_4x32, res1_4x32);
637
638
0
    res0_4x64 = _mm_unpacklo_epi32(res0_4x32, zeros);
639
0
    res1_4x64 = _mm_unpackhi_epi32(res0_4x32, zeros);
640
0
    res2_4x64 = _mm_unpacklo_epi32(res1_4x32, zeros);
641
0
    res3_4x64 = _mm_unpackhi_epi32(res1_4x32, zeros);
642
643
0
    square_result = _mm_add_epi64(
644
0
        square_result,
645
0
        _mm_add_epi64(
646
0
            _mm_add_epi64(_mm_add_epi64(res0_4x64, res1_4x64), res2_4x64),
647
0
            res3_4x64));
648
0
  }
649
650
0
  const __m128i sum_1x64 =
651
0
      _mm_add_epi64(square_result, _mm_srli_si128(square_result, 8));
652
0
  xx_storel_64(&sum, sum_1x64);
653
0
  return sum;
654
0
}
655
656
static uint64_t mse_8xh_16bit_highbd_sse2(uint16_t *dst, int dstride,
657
0
                                          uint16_t *src, int sstride, int h) {
658
0
  uint64_t sum = 0;
659
0
  __m128i src_8x16;
660
0
  __m128i dst_8x16;
661
0
  __m128i res0_4x32, res1_4x32, res0_4x64, res1_4x64, res2_4x64, res3_4x64;
662
0
  __m128i sub_result_8x16;
663
0
  const __m128i zeros = _mm_setzero_si128();
664
0
  __m128i square_result = _mm_setzero_si128();
665
666
0
  for (int i = 0; i < h; i++) {
667
0
    dst_8x16 = _mm_loadu_si128((__m128i *)&dst[i * dstride]);
668
0
    src_8x16 = _mm_loadu_si128((__m128i *)&src[i * sstride]);
669
670
0
    sub_result_8x16 = _mm_sub_epi16(src_8x16, dst_8x16);
671
672
0
    res0_4x32 = _mm_unpacklo_epi16(sub_result_8x16, zeros);
673
0
    res1_4x32 = _mm_unpackhi_epi16(sub_result_8x16, zeros);
674
675
0
    res0_4x32 = _mm_madd_epi16(res0_4x32, res0_4x32);
676
0
    res1_4x32 = _mm_madd_epi16(res1_4x32, res1_4x32);
677
678
0
    res0_4x64 = _mm_unpacklo_epi32(res0_4x32, zeros);
679
0
    res1_4x64 = _mm_unpackhi_epi32(res0_4x32, zeros);
680
0
    res2_4x64 = _mm_unpacklo_epi32(res1_4x32, zeros);
681
0
    res3_4x64 = _mm_unpackhi_epi32(res1_4x32, zeros);
682
683
0
    square_result = _mm_add_epi64(
684
0
        square_result,
685
0
        _mm_add_epi64(
686
0
            _mm_add_epi64(_mm_add_epi64(res0_4x64, res1_4x64), res2_4x64),
687
0
            res3_4x64));
688
0
  }
689
690
0
  const __m128i sum_1x64 =
691
0
      _mm_add_epi64(square_result, _mm_srli_si128(square_result, 8));
692
0
  xx_storel_64(&sum, sum_1x64);
693
0
  return sum;
694
0
}
695
696
uint64_t aom_mse_wxh_16bit_highbd_sse2(uint16_t *dst, int dstride,
697
                                       uint16_t *src, int sstride, int w,
698
0
                                       int h) {
699
0
  assert((w == 8 || w == 4) && (h == 8 || h == 4) &&
700
0
         "w=8/4 and h=8/4 must satisfy");
701
0
  switch (w) {
702
0
    case 4: return mse_4xh_16bit_highbd_sse2(dst, dstride, src, sstride, h);
703
0
    case 8: return mse_8xh_16bit_highbd_sse2(dst, dstride, src, sstride, h);
704
0
    default: assert(0 && "unsupported width"); return -1;
705
0
  }
706
0
}