Coverage Report

Created: 2026-09-13 06:34

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/work/libde265/libde265/x86/sse-motion.cc
Line
Count
Source
1
/*
2
 * H.265 video codec.
3
 * Copyright (c) 2013 openHEVC contributors
4
 * Copyright (c) 2013-2014 struktur AG, Dirk Farin <farin@struktur.de>
5
 *
6
 * This file is part of libde265.
7
 *
8
 * libde265 is free software: you can redistribute it and/or modify
9
 * it under the terms of the GNU Lesser General Public License as
10
 * published by the Free Software Foundation, either version 3 of
11
 * the License, or (at your option) any later version.
12
 *
13
 * libde265 is distributed in the hope that it will be useful,
14
 * but WITHOUT ANY WARRANTY; without even the implied warranty of
15
 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
16
 * GNU Lesser General Public License for more details.
17
 *
18
 * You should have received a copy of the GNU Lesser General Public License
19
 * along with libde265.  If not, see <http://www.gnu.org/licenses/>.
20
 */
21
22
#ifdef HAVE_CONFIG_H
23
#include "config.h"
24
#endif
25
26
#include <stdio.h>
27
#include <emmintrin.h>
28
#include <tmmintrin.h> // SSSE3
29
#if HAVE_SSE4_1
30
#include <smmintrin.h>
31
#endif
32
33
#include "sse-motion.h"
34
#include "libde265/util.h"
35
36
37
ALIGNED_16(const int8_t) epel_filters[7][16] = {
38
  { -2,  58,  10,  -2,-2,  58,  10,  -2,-2,  58,  10,  -2,-2,  58,  10,  -2 },
39
  { -4,  54,  16,  -2,-4,  54,  16,  -2,-4,  54,  16,  -2,-4,  54,  16,  -2 },
40
  { -6,  46,  28,  -4,-6,  46,  28,  -4,-6,  46,  28,  -4,-6,  46,  28,  -4 },
41
  { -4,  36,  36,  -4,-4,  36,  36,  -4,-4,  36,  36,  -4,-4,  36,  36,  -4 },
42
  { -4,  28,  46,  -6,-4,  28,  46,  -6,-4,  28,  46,  -6,-4,  28,  46,  -6 },
43
  { -2,  16,  54,  -4,-2,  16,  54,  -4,-2,  16,  54,  -4,-2,  16,  54,  -4 },
44
  { -2,  10,  58,  -2,-2,  10,  58,  -2,-2,  10,  58,  -2,-2,  10,  58,  -2 },
45
};
46
47
static const uint8_t qpel_extra_before[4] = { 0, 3, 3, 2 };
48
//static const uint8_t qpel_extra_after[4] = { 0, 3, 4, 4 };
49
static const uint8_t qpel_extra[4] = { 0, 6, 7, 6 };
50
51
static const int epel_extra_before = 1;
52
//static const int epel_extra_after = 2;
53
static const int epel_extra = 3;
54
55
203M
#define MAX_PB_SIZE 64
56
57
#define MASKMOVE 0
58
59
void print128(const char* prefix, __m128i r)
60
0
{
61
0
  unsigned char buf[16];
62
63
0
  *(__m128i*)buf = r;
64
65
0
  printf("%s ",prefix);
66
0
  for (int i=0;i<16;i++)
67
0
    {
68
0
      if (i>0) { printf(":"); }
69
0
      printf("%02x", buf[i]);
70
0
    }
71
72
0
  printf("\n");
73
0
}
74
75
76
void printm32(const char* prefix, unsigned char* p)
77
0
{
78
0
  printf("%s ",prefix);
79
80
0
  for (int i=0;i<4;i++)
81
0
    {
82
0
      if (i>0) { printf(":"); }
83
0
      printf("%02x", p[i]);
84
0
    }
85
86
0
  printf("\n");
87
0
}
88
89
90
4.97M
#define BIT_DEPTH 8
91
92
void ff_hevc_put_unweighted_pred_8_sse(uint8_t *_dst, ptrdiff_t dststride,
93
                                       const int16_t *src, ptrdiff_t srcstride,
94
12.4M
                                       int width, int height) {
95
12.4M
    int x, y;
96
12.4M
    uint8_t *dst = (uint8_t*) _dst;
97
12.4M
    __m128i r0, r1, f0;
98
99
12.4M
    f0 = _mm_set1_epi16(32);
100
101
102
12.4M
    if(!(width & 15))
103
568k
    {
104
9.64M
        for (y = 0; y < height; y++) {
105
20.1M
                    for (x = 0; x < width; x += 16) {
106
11.1M
                        r0 = _mm_load_si128((__m128i *) (src+x));
107
108
11.1M
                        r1 = _mm_load_si128((__m128i *) (src+x + 8));
109
11.1M
                        r0 = _mm_adds_epi16(r0, f0);
110
111
11.1M
                        r1 = _mm_adds_epi16(r1, f0);
112
11.1M
                        r0 = _mm_srai_epi16(r0, 6);
113
11.1M
                        r1 = _mm_srai_epi16(r1, 6);
114
11.1M
                        r0 = _mm_packus_epi16(r0, r1);
115
116
11.1M
                        _mm_storeu_si128((__m128i *) (dst+x), r0);
117
11.1M
                    }
118
9.07M
                    dst += dststride;
119
9.07M
                    src += srcstride;
120
9.07M
                }
121
11.9M
    }else if(!(width & 7))
122
3.01M
    {
123
26.0M
        for (y = 0; y < height; y++) {
124
46.3M
            for (x = 0; x < width; x += 8) {
125
23.3M
                    r0 = _mm_load_si128((__m128i *) (src+x));
126
127
23.3M
                    r0 = _mm_adds_epi16(r0, f0);
128
129
23.3M
                    r0 = _mm_srai_epi16(r0, 6);
130
23.3M
                    r0 = _mm_packus_epi16(r0, r0);
131
132
23.3M
                    _mm_storel_epi64((__m128i *) (dst+x), r0);
133
23.3M
            }
134
23.0M
                    dst += dststride;
135
23.0M
                    src += srcstride;
136
23.0M
                }
137
8.90M
    }else if(!(width & 3)){
138
39.9M
        for (y = 0; y < height; y++) {
139
66.2M
                    for(x = 0;x < width; x+=4){
140
33.5M
                    r0 = _mm_loadl_epi64((__m128i *) (src+x));
141
33.5M
                    r0 = _mm_adds_epi16(r0, f0);
142
143
33.5M
                    r0 = _mm_srai_epi16(r0, 6);
144
33.5M
                    r0 = _mm_packus_epi16(r0, r0);
145
#if MASKMOVE
146
                    _mm_maskmoveu_si128(r0,_mm_set_epi8(0,0,0,0,0,0,0,0,0,0,0,0,-1,-1,-1,-1),(char *) (dst+x));
147
#else
148
                    //r0 = _mm_shuffle_epi32 (r0, 0x00);
149
33.5M
                    *((uint32_t*)(dst+x)) = _mm_cvtsi128_si32(r0);
150
33.5M
#endif
151
33.5M
                    }
152
32.7M
                    dst += dststride;
153
32.7M
                    src += srcstride;
154
32.7M
                }
155
7.24M
    }else{
156
10.7M
        for (y = 0; y < height; y++) {
157
18.7M
                    for(x = 0;x < width; x+=2){
158
9.63M
                    r0 = _mm_loadl_epi64((__m128i *) (src+x));
159
9.63M
                    r0 = _mm_adds_epi16(r0, f0);
160
161
9.63M
                    r0 = _mm_srai_epi16(r0, 6);
162
9.63M
                    r0 = _mm_packus_epi16(r0, r0);
163
#if MASKMOVE
164
                    _mm_maskmoveu_si128(r0,_mm_set_epi8(0,0,0,0,0,0,0,0,0,0,0,0,0,0,-1,-1),(char *) (dst+x));
165
#else
166
9.63M
                    *((uint16_t*)(dst+x)) = _mm_cvtsi128_si32(r0);
167
9.63M
#endif
168
9.63M
                    }
169
9.11M
                    dst += dststride;
170
9.11M
                    src += srcstride;
171
9.11M
                }
172
1.65M
    }
173
174
12.4M
}
175
176
void ff_hevc_put_unweighted_pred_sse(uint8_t *_dst, ptrdiff_t _dststride,
177
                                     const int16_t *src, ptrdiff_t srcstride,
178
0
                                     int width, int height) {
179
0
    int x, y;
180
0
    uint8_t *dst = (uint8_t*) _dst;
181
0
    ptrdiff_t dststride = _dststride / sizeof(uint8_t);
182
0
    __m128i r0, r1, f0;
183
0
    int shift = 14 - BIT_DEPTH;
184
0
#if BIT_DEPTH < 14
185
0
    int16_t offset = 1 << (shift - 1);
186
#else
187
    int16_t offset = 0;
188
189
#endif
190
0
    f0 = _mm_set1_epi16(offset);
191
192
0
    for (y = 0; y < height; y++) {
193
0
        for (x = 0; x < width; x += 16) {
194
0
            r0 = _mm_load_si128((__m128i *) &src[x]);
195
196
0
            r1 = _mm_load_si128((__m128i *) &src[x + 8]);
197
0
            r0 = _mm_adds_epi16(r0, f0);
198
199
0
            r1 = _mm_adds_epi16(r1, f0);
200
0
            r0 = _mm_srai_epi16(r0, shift);
201
0
            r1 = _mm_srai_epi16(r1, shift);
202
0
            r0 = _mm_packus_epi16(r0, r1);
203
204
0
            _mm_storeu_si128((__m128i *) &dst[x], r0);
205
0
        }
206
0
        dst += dststride;
207
0
        src += srcstride;
208
0
    }
209
0
}
210
211
void ff_hevc_put_weighted_pred_avg_8_sse(uint8_t *_dst, ptrdiff_t dststride,
212
                                         const int16_t *src1, const int16_t *src2,
213
                                         ptrdiff_t srcstride, int width,
214
2.20M
                                         int height) {
215
2.20M
    int x, y;
216
2.20M
    uint8_t *dst = (uint8_t*) _dst;
217
2.20M
    __m128i r0, r1, f0, r2, r3;
218
219
2.20M
    f0 = _mm_set1_epi16(64);
220
2.20M
    if(!(width & 15)){
221
4.80M
        for (y = 0; y < height; y++) {
222
223
10.3M
            for (x = 0; x < width; x += 16) {
224
5.84M
                r0 = _mm_load_si128((__m128i *) &src1[x]);
225
5.84M
                r1 = _mm_load_si128((__m128i *) &src1[x + 8]);
226
5.84M
                r2 = _mm_load_si128((__m128i *) &src2[x]);
227
5.84M
                r3 = _mm_load_si128((__m128i *) &src2[x + 8]);
228
229
5.84M
                r0 = _mm_adds_epi16(r0, f0);
230
5.84M
                r1 = _mm_adds_epi16(r1, f0);
231
5.84M
                r0 = _mm_adds_epi16(r0, r2);
232
5.84M
                r1 = _mm_adds_epi16(r1, r3);
233
5.84M
                r0 = _mm_srai_epi16(r0, 7);
234
5.84M
                r1 = _mm_srai_epi16(r1, 7);
235
5.84M
                r0 = _mm_packus_epi16(r0, r1);
236
237
5.84M
                _mm_storeu_si128((__m128i *) (dst + x), r0);
238
5.84M
            }
239
4.52M
            dst += dststride;
240
4.52M
            src1 += srcstride;
241
4.52M
            src2 += srcstride;
242
4.52M
        }
243
1.92M
    }else if(!(width & 7)){
244
8.36M
        for (y = 0; y < height; y++) {
245
15.1M
            for(x=0;x<width;x+=8){
246
7.65M
                r0 = _mm_load_si128((__m128i *) (src1+x));
247
7.65M
                r2 = _mm_load_si128((__m128i *) (src2+x));
248
249
7.65M
                r0 = _mm_adds_epi16(r0, f0);
250
7.65M
                r0 = _mm_adds_epi16(r0, r2);
251
7.65M
                r0 = _mm_srai_epi16(r0, 7);
252
7.65M
                r0 = _mm_packus_epi16(r0, r0);
253
254
7.65M
                _mm_storel_epi64((__m128i *) (dst+x), r0);
255
7.65M
            }
256
7.53M
            dst += dststride;
257
7.53M
            src1 += srcstride;
258
7.53M
            src2 += srcstride;
259
7.53M
        }
260
1.08M
    }else if(!(width & 3)){
261
#if MASKMOVE
262
      r1= _mm_set_epi8(0,0,0,0,0,0,0,0,0,0,0,0,-1,-1,-1,-1);
263
#endif
264
7.10M
        for (y = 0; y < height; y++) {
265
266
12.2M
            for(x=0;x<width;x+=4)
267
6.23M
            {
268
6.23M
                r0 = _mm_loadl_epi64((__m128i *) (src1+x));
269
6.23M
                r2 = _mm_loadl_epi64((__m128i *) (src2+x));
270
271
6.23M
                r0 = _mm_adds_epi16(r0, f0);
272
6.23M
                r0 = _mm_adds_epi16(r0, r2);
273
6.23M
                r0 = _mm_srai_epi16(r0, 7);
274
6.23M
                r0 = _mm_packus_epi16(r0, r0);
275
276
#if MASKMOVE
277
                _mm_maskmoveu_si128(r0,r1,(char *) (dst+x));
278
#else
279
6.23M
                *((uint32_t*)(dst+x)) = _mm_cvtsi128_si32(r0);
280
6.23M
#endif
281
6.23M
            }
282
6.02M
            dst += dststride;
283
6.02M
            src1 += srcstride;
284
6.02M
            src2 += srcstride;
285
6.02M
        }
286
1.07M
    }else{
287
#if MASKMOVE
288
      r1= _mm_set_epi8(0,0,0,0,0,0,0,0,0,0,0,0,0,0,-1,-1);
289
#endif
290
130k
        for (y = 0; y < height; y++) {
291
353k
                    for(x=0;x<width;x+=2)
292
235k
                    {
293
235k
                        r0 = _mm_loadl_epi64((__m128i *) (src1+x));
294
235k
                        r2 = _mm_loadl_epi64((__m128i *) (src2+x));
295
296
235k
                        r0 = _mm_adds_epi16(r0, f0);
297
235k
                        r0 = _mm_adds_epi16(r0, r2);
298
235k
                        r0 = _mm_srai_epi16(r0, 7);
299
235k
                        r0 = _mm_packus_epi16(r0, r0);
300
301
#if MASKMOVE
302
                        _mm_maskmoveu_si128(r0,r1,(char *) (dst+x));
303
#else
304
235k
                        *((uint16_t*)(dst+x)) = _mm_cvtsi128_si32(r0);
305
235k
#endif
306
235k
                    }
307
117k
                    dst += dststride;
308
117k
                    src1 += srcstride;
309
117k
                    src2 += srcstride;
310
117k
                }
311
12.2k
    }
312
313
314
2.20M
}
315
316
void ff_hevc_put_weighted_pred_avg_sse(uint8_t *_dst, ptrdiff_t _dststride,
317
                                       const int16_t *src1, const int16_t *src2,
318
                                       ptrdiff_t srcstride, int width,
319
0
                                       int height) {
320
0
    int x, y;
321
0
    uint8_t *dst = (uint8_t*) _dst;
322
0
    ptrdiff_t dststride = _dststride / sizeof(uint8_t);
323
0
    __m128i r0, r1, f0, r2, r3;
324
0
    int shift = 14 + 1 - BIT_DEPTH;
325
0
#if BIT_DEPTH < 14
326
0
    int offset = 1 << (shift - 1);
327
#else
328
    int offset = 0;
329
#endif
330
0
    f0 = _mm_set1_epi16(offset);
331
0
    for (y = 0; y < height; y++) {
332
333
0
        for (x = 0; x < width; x += 16) {
334
0
            r0 = _mm_load_si128((__m128i *) &src1[x]);
335
0
            r1 = _mm_load_si128((__m128i *) &src1[x + 8]);
336
0
            r2 = _mm_load_si128((__m128i *) &src2[x]);
337
0
            r3 = _mm_load_si128((__m128i *) &src2[x + 8]);
338
339
0
            r0 = _mm_adds_epi16(r0, f0);
340
0
            r1 = _mm_adds_epi16(r1, f0);
341
0
            r0 = _mm_adds_epi16(r0, r2);
342
0
            r1 = _mm_adds_epi16(r1, r3);
343
0
            r0 = _mm_srai_epi16(r0, shift);
344
0
            r1 = _mm_srai_epi16(r1, shift);
345
0
            r0 = _mm_packus_epi16(r0, r1);
346
347
0
            _mm_storeu_si128((__m128i *) (dst + x), r0);
348
0
        }
349
0
        dst += dststride;
350
0
        src1 += srcstride;
351
0
        src2 += srcstride;
352
0
    }
353
0
}
354
355
#if 0
356
void ff_hevc_weighted_pred_8_sse4(uint8_t denom, int16_t wlxFlag, int16_t olxFlag,
357
                                  uint8_t *_dst, ptrdiff_t _dststride,
358
                                  const int16_t *src, ptrdiff_t srcstride,
359
                                  int width, int height) {
360
361
    int log2Wd;
362
    int x, y;
363
364
    uint8_t *dst = (uint8_t*) _dst;
365
    ptrdiff_t dststride = _dststride / sizeof(uint8_t);
366
    __m128i x0, x1, x2, x3, c0, add, add2;
367
368
    log2Wd = denom + 14 - BIT_DEPTH;
369
370
    add = _mm_set1_epi32(olxFlag * (1 << (BIT_DEPTH - 8)));
371
    add2 = _mm_set1_epi32(1 << (log2Wd - 1));
372
    c0 = _mm_set1_epi16(wlxFlag);
373
    if (log2Wd >= 1){
374
        if(!(width & 15)){
375
            for (y = 0; y < height; y++) {
376
                for (x = 0; x < width; x += 16) {
377
                    x0 = _mm_load_si128((__m128i *) &src[x]);
378
                    x2 = _mm_load_si128((__m128i *) &src[x + 8]);
379
                    x1 = _mm_unpackhi_epi16(_mm_mullo_epi16(x0, c0),
380
                            _mm_mulhi_epi16(x0, c0));
381
                    x3 = _mm_unpackhi_epi16(_mm_mullo_epi16(x2, c0),
382
                            _mm_mulhi_epi16(x2, c0));
383
                    x0 = _mm_unpacklo_epi16(_mm_mullo_epi16(x0, c0),
384
                            _mm_mulhi_epi16(x0, c0));
385
                    x2 = _mm_unpacklo_epi16(_mm_mullo_epi16(x2, c0),
386
                            _mm_mulhi_epi16(x2, c0));
387
                    x0 = _mm_add_epi32(x0, add2);
388
                    x1 = _mm_add_epi32(x1, add2);
389
                    x2 = _mm_add_epi32(x2, add2);
390
                    x3 = _mm_add_epi32(x3, add2);
391
                    x0 = _mm_srai_epi32(x0, log2Wd);
392
                    x1 = _mm_srai_epi32(x1, log2Wd);
393
                    x2 = _mm_srai_epi32(x2, log2Wd);
394
                    x3 = _mm_srai_epi32(x3, log2Wd);
395
                    x0 = _mm_add_epi32(x0, add);
396
                    x1 = _mm_add_epi32(x1, add);
397
                    x2 = _mm_add_epi32(x2, add);
398
                    x3 = _mm_add_epi32(x3, add);
399
                    x0 = _mm_packus_epi32(x0, x1);
400
                    x2 = _mm_packus_epi32(x2, x3);
401
                    x0 = _mm_packus_epi16(x0, x2);
402
403
                    _mm_storeu_si128((__m128i *) (dst + x), x0);
404
405
                }
406
                dst += dststride;
407
                src += srcstride;
408
            }
409
        }else if(!(width & 7)){
410
            for (y = 0; y < height; y++) {
411
                for(x=0;x<width;x+=8){
412
                    x0 = _mm_load_si128((__m128i *) (src+x));
413
                    x1 = _mm_unpackhi_epi16(_mm_mullo_epi16(x0, c0),
414
                            _mm_mulhi_epi16(x0, c0));
415
416
                    x0 = _mm_unpacklo_epi16(_mm_mullo_epi16(x0, c0),
417
                            _mm_mulhi_epi16(x0, c0));
418
419
                    x0 = _mm_add_epi32(x0, add2);
420
                    x1 = _mm_add_epi32(x1, add2);
421
422
                    x0 = _mm_srai_epi32(x0, log2Wd);
423
                    x1 = _mm_srai_epi32(x1, log2Wd);
424
425
                    x0 = _mm_add_epi32(x0, add);
426
                    x1 = _mm_add_epi32(x1, add);
427
428
                    x0 = _mm_packus_epi32(x0, x1);
429
                    x0 = _mm_packus_epi16(x0, x0);
430
431
                    _mm_storel_epi64((__m128i *) (dst+x), x0);
432
433
                }
434
                dst += dststride;
435
                src += srcstride;
436
            }
437
        }else if(!(width & 3)){
438
            for (y = 0; y < height; y++) {
439
                for(x=0;x<width;x+=4){
440
                    x0 = _mm_loadl_epi64((__m128i *)(src+x));
441
                    x1 = _mm_unpackhi_epi16(_mm_mullo_epi16(x0, c0),
442
                            _mm_mulhi_epi16(x0, c0));
443
                    x0 = _mm_unpacklo_epi16(_mm_mullo_epi16(x0, c0),
444
                            _mm_mulhi_epi16(x0, c0));
445
446
                    x0 = _mm_add_epi32(x0, add2);
447
                    x1 = _mm_add_epi32(x1, add2);
448
                    x0 = _mm_srai_epi32(x0, log2Wd);
449
                    x1 = _mm_srai_epi32(x1, log2Wd);
450
                    x0 = _mm_add_epi32(x0, add);
451
                    x1 = _mm_add_epi32(x1, add);
452
                    x0 = _mm_packus_epi32(x0, x1);
453
                    x0 = _mm_packus_epi16(x0, x0);
454
455
                    _mm_maskmoveu_si128(x0,_mm_set_epi8(0,0,0,0,0,0,0,0,0,0,0,0,-1,-1,-1,-1),(char *) (dst+x));
456
                    // _mm_storeu_si128((__m128i *) (dst + x), x0);
457
                }
458
                dst += dststride;
459
                src += srcstride;
460
            }
461
        }else{
462
            for (y = 0; y < height; y++) {
463
                for(x=0;x<width;x+=2){
464
                    x0 = _mm_loadl_epi64((__m128i *)(src+x));
465
                    x1 = _mm_unpackhi_epi16(_mm_mullo_epi16(x0, c0),
466
                            _mm_mulhi_epi16(x0, c0));
467
                    x0 = _mm_unpacklo_epi16(_mm_mullo_epi16(x0, c0),
468
                            _mm_mulhi_epi16(x0, c0));
469
470
                    x0 = _mm_add_epi32(x0, add2);
471
                    x1 = _mm_add_epi32(x1, add2);
472
                    x0 = _mm_srai_epi32(x0, log2Wd);
473
                    x1 = _mm_srai_epi32(x1, log2Wd);
474
                    x0 = _mm_add_epi32(x0, add);
475
                    x1 = _mm_add_epi32(x1, add);
476
                    x0 = _mm_packus_epi32(x0, x1);
477
                    x0 = _mm_packus_epi16(x0, x0);
478
479
                    _mm_maskmoveu_si128(x0,_mm_set_epi8(0,0,0,0,0,0,0,0,0,0,0,0,0,0,-1,-1),(char *) (dst+x));
480
                    // _mm_storeu_si128((__m128i *) (dst + x), x0);
481
                }
482
                dst += dststride;
483
                src += srcstride;
484
            }
485
        }
486
    }else{
487
        if(!(width & 15)){
488
            for (y = 0; y < height; y++) {
489
                for (x = 0; x < width; x += 16) {
490
491
                    x0 = _mm_load_si128((__m128i *) &src[x]);
492
                    x2 = _mm_load_si128((__m128i *) &src[x + 8]);
493
                    x1 = _mm_unpackhi_epi16(_mm_mullo_epi16(x0, c0),
494
                            _mm_mulhi_epi16(x0, c0));
495
                    x3 = _mm_unpackhi_epi16(_mm_mullo_epi16(x2, c0),
496
                            _mm_mulhi_epi16(x2, c0));
497
                    x0 = _mm_unpacklo_epi16(_mm_mullo_epi16(x0, c0),
498
                            _mm_mulhi_epi16(x0, c0));
499
                    x2 = _mm_unpacklo_epi16(_mm_mullo_epi16(x2, c0),
500
                            _mm_mulhi_epi16(x2, c0));
501
502
                    x0 = _mm_add_epi32(x0, add2);
503
                    x1 = _mm_add_epi32(x1, add2);
504
                    x2 = _mm_add_epi32(x2, add2);
505
                    x3 = _mm_add_epi32(x3, add2);
506
507
                    x0 = _mm_packus_epi32(x0, x1);
508
                    x2 = _mm_packus_epi32(x2, x3);
509
                    x0 = _mm_packus_epi16(x0, x2);
510
511
                    _mm_storeu_si128((__m128i *) (dst + x), x0);
512
513
                }
514
                dst += dststride;
515
                src += srcstride;
516
            }
517
        }else if(!(width & 7)){
518
            for (y = 0; y < height; y++) {
519
                for(x=0;x<width;x+=8){
520
                    x0 = _mm_load_si128((__m128i *) (src+x));
521
                    x1 = _mm_unpackhi_epi16(_mm_mullo_epi16(x0, c0),
522
                            _mm_mulhi_epi16(x0, c0));
523
524
                    x0 = _mm_unpacklo_epi16(_mm_mullo_epi16(x0, c0),
525
                            _mm_mulhi_epi16(x0, c0));
526
527
528
                    x0 = _mm_add_epi32(x0, add2);
529
                    x1 = _mm_add_epi32(x1, add2);
530
531
                    x0 = _mm_packus_epi32(x0, x1);
532
                    x0 = _mm_packus_epi16(x0, x0);
533
534
                    _mm_storeu_si128((__m128i *) (dst+x), x0);
535
                }
536
537
                dst += dststride;
538
                src += srcstride;
539
            }
540
        }else if(!(width & 3)){
541
            for (y = 0; y < height; y++) {
542
                for(x=0;x<width;x+=4){
543
                    x0 = _mm_loadl_epi64((__m128i *) (src+x));
544
                    x1 = _mm_unpackhi_epi16(_mm_mullo_epi16(x0, c0),
545
                            _mm_mulhi_epi16(x0, c0));
546
547
                    x0 = _mm_unpacklo_epi16(_mm_mullo_epi16(x0, c0),
548
                            _mm_mulhi_epi16(x0, c0));
549
550
551
                    x0 = _mm_add_epi32(x0, add2);
552
                    x1 = _mm_add_epi32(x1, add2);
553
554
555
                    x0 = _mm_packus_epi32(x0, x1);
556
                    x0 = _mm_packus_epi16(x0, x0);
557
558
559
                    _mm_maskmoveu_si128(x0,_mm_set_epi8(0,0,0,0,0,0,0,0,0,0,0,0,-1,-1,-1,-1),(char *) (dst+x));
560
                }
561
                dst += dststride;
562
                src += srcstride;
563
            }
564
        }else{
565
            for (y = 0; y < height; y++) {
566
                for(x=0;x<width;x+=2){
567
                    x0 = _mm_loadl_epi64((__m128i *) (src+x));
568
                    x1 = _mm_unpackhi_epi16(_mm_mullo_epi16(x0, c0),
569
                            _mm_mulhi_epi16(x0, c0));
570
571
                    x0 = _mm_unpacklo_epi16(_mm_mullo_epi16(x0, c0),
572
                            _mm_mulhi_epi16(x0, c0));
573
574
575
                    x0 = _mm_add_epi32(x0, add2);
576
                    x1 = _mm_add_epi32(x1, add2);
577
578
579
                    x0 = _mm_packus_epi32(x0, x1);
580
                    x0 = _mm_packus_epi16(x0, x0);
581
582
583
                    _mm_maskmoveu_si128(x0,_mm_set_epi8(0,0,0,0,0,0,0,0,0,0,0,0,0,0,-1,-1),(char *) (dst+x));
584
                }
585
                dst += dststride;
586
                src += srcstride;
587
            }
588
589
        }
590
591
    }
592
593
}
594
#endif
595
596
597
#if 0
598
void ff_hevc_weighted_pred_sse(uint8_t denom, int16_t wlxFlag, int16_t olxFlag,
599
                               uint8_t *_dst, ptrdiff_t _dststride,
600
                               const int16_t *src, ptrdiff_t srcstride,
601
                               int width, int height) {
602
603
    int log2Wd;
604
    int x, y;
605
606
    uint8_t *dst = (uint8_t*) _dst;
607
    ptrdiff_t dststride = _dststride / sizeof(uint8_t);
608
    __m128i x0, x1, x2, x3, c0, add, add2;
609
610
    log2Wd = denom + 14 - BIT_DEPTH;
611
612
    add = _mm_set1_epi32(olxFlag * (1 << (BIT_DEPTH - 8)));
613
    add2 = _mm_set1_epi32(1 << (log2Wd - 1));
614
    c0 = _mm_set1_epi16(wlxFlag);
615
    if (log2Wd >= 1)
616
        for (y = 0; y < height; y++) {
617
            for (x = 0; x < width; x += 16) {
618
                x0 = _mm_load_si128((__m128i *) &src[x]);
619
                x2 = _mm_load_si128((__m128i *) &src[x + 8]);
620
                x1 = _mm_unpackhi_epi16(_mm_mullo_epi16(x0, c0),
621
                        _mm_mulhi_epi16(x0, c0));
622
                x3 = _mm_unpackhi_epi16(_mm_mullo_epi16(x2, c0),
623
                        _mm_mulhi_epi16(x2, c0));
624
                x0 = _mm_unpacklo_epi16(_mm_mullo_epi16(x0, c0),
625
                        _mm_mulhi_epi16(x0, c0));
626
                x2 = _mm_unpacklo_epi16(_mm_mullo_epi16(x2, c0),
627
                        _mm_mulhi_epi16(x2, c0));
628
                x0 = _mm_add_epi32(x0, add2);
629
                x1 = _mm_add_epi32(x1, add2);
630
                x2 = _mm_add_epi32(x2, add2);
631
                x3 = _mm_add_epi32(x3, add2);
632
                x0 = _mm_srai_epi32(x0, log2Wd);
633
                x1 = _mm_srai_epi32(x1, log2Wd);
634
                x2 = _mm_srai_epi32(x2, log2Wd);
635
                x3 = _mm_srai_epi32(x3, log2Wd);
636
                x0 = _mm_add_epi32(x0, add);
637
                x1 = _mm_add_epi32(x1, add);
638
                x2 = _mm_add_epi32(x2, add);
639
                x3 = _mm_add_epi32(x3, add);
640
                x0 = _mm_packus_epi32(x0, x1);
641
                x2 = _mm_packus_epi32(x2, x3);
642
                x0 = _mm_packus_epi16(x0, x2);
643
644
                _mm_storeu_si128((__m128i *) (dst + x), x0);
645
646
            }
647
            dst += dststride;
648
            src += srcstride;
649
        }
650
    else
651
        for (y = 0; y < height; y++) {
652
            for (x = 0; x < width; x += 16) {
653
654
                x0 = _mm_load_si128((__m128i *) &src[x]);
655
                x2 = _mm_load_si128((__m128i *) &src[x + 8]);
656
                x1 = _mm_unpackhi_epi16(_mm_mullo_epi16(x0, c0),
657
                        _mm_mulhi_epi16(x0, c0));
658
                x3 = _mm_unpackhi_epi16(_mm_mullo_epi16(x2, c0),
659
                        _mm_mulhi_epi16(x2, c0));
660
                x0 = _mm_unpacklo_epi16(_mm_mullo_epi16(x0, c0),
661
                        _mm_mulhi_epi16(x0, c0));
662
                x2 = _mm_unpacklo_epi16(_mm_mullo_epi16(x2, c0),
663
                        _mm_mulhi_epi16(x2, c0));
664
665
                x0 = _mm_add_epi32(x0, add2);
666
                x1 = _mm_add_epi32(x1, add2);
667
                x2 = _mm_add_epi32(x2, add2);
668
                x3 = _mm_add_epi32(x3, add2);
669
670
                x0 = _mm_packus_epi32(x0, x1);
671
                x2 = _mm_packus_epi32(x2, x3);
672
                x0 = _mm_packus_epi16(x0, x2);
673
674
                _mm_storeu_si128((__m128i *) (dst + x), x0);
675
676
            }
677
            dst += dststride;
678
            src += srcstride;
679
        }
680
}
681
#endif
682
683
#if HAVE_SSE4_1
684
void ff_hevc_weighted_pred_avg_8_sse4(uint8_t denom, int16_t wl0Flag,
685
                                      int16_t wl1Flag, int16_t ol0Flag, int16_t ol1Flag,
686
                                      uint8_t *_dst, ptrdiff_t _dststride,
687
                                      const int16_t *src1, const int16_t *src2, ptrdiff_t srcstride,
688
0
                                      int width, int height) {
689
0
    int shift, shift2;
690
0
    int log2Wd;
691
0
    int o0;
692
0
    int o1;
693
0
    int x, y;
694
0
    uint8_t *dst = (uint8_t*) _dst;
695
0
    ptrdiff_t dststride = _dststride / sizeof(uint8_t);
696
0
    __m128i x0, x1, x2, x3, r0, r1, r2, r3, c0, c1, c2;
697
0
    shift = 14 - BIT_DEPTH;
698
0
    log2Wd = denom + shift;
699
700
0
    o0 = (ol0Flag) * (1 << (BIT_DEPTH - 8));
701
0
    o1 = (ol1Flag) * (1 << (BIT_DEPTH - 8));
702
0
    shift2 = (log2Wd + 1);
703
0
    c0 = _mm_set1_epi16(wl0Flag);
704
0
    c1 = _mm_set1_epi16(wl1Flag);
705
0
    c2 = _mm_set1_epi32((o0 + o1 + 1) << log2Wd);
706
707
0
    if(!(width & 15)){
708
0
        for (y = 0; y < height; y++) {
709
0
                   for (x = 0; x < width; x += 16) {
710
0
                       x0 = _mm_load_si128((__m128i *) &src1[x]);
711
0
                       x1 = _mm_load_si128((__m128i *) &src1[x + 8]);
712
0
                       x2 = _mm_load_si128((__m128i *) &src2[x]);
713
0
                       x3 = _mm_load_si128((__m128i *) &src2[x + 8]);
714
715
0
                       r0 = _mm_unpacklo_epi16(_mm_mullo_epi16(x0, c0),
716
0
                               _mm_mulhi_epi16(x0, c0));
717
0
                       r1 = _mm_unpacklo_epi16(_mm_mullo_epi16(x1, c0),
718
0
                               _mm_mulhi_epi16(x1, c0));
719
0
                       r2 = _mm_unpacklo_epi16(_mm_mullo_epi16(x2, c1),
720
0
                               _mm_mulhi_epi16(x2, c1));
721
0
                       r3 = _mm_unpacklo_epi16(_mm_mullo_epi16(x3, c1),
722
0
                               _mm_mulhi_epi16(x3, c1));
723
0
                       x0 = _mm_unpackhi_epi16(_mm_mullo_epi16(x0, c0),
724
0
                               _mm_mulhi_epi16(x0, c0));
725
0
                       x1 = _mm_unpackhi_epi16(_mm_mullo_epi16(x1, c0),
726
0
                               _mm_mulhi_epi16(x1, c0));
727
0
                       x2 = _mm_unpackhi_epi16(_mm_mullo_epi16(x2, c1),
728
0
                               _mm_mulhi_epi16(x2, c1));
729
0
                       x3 = _mm_unpackhi_epi16(_mm_mullo_epi16(x3, c1),
730
0
                               _mm_mulhi_epi16(x3, c1));
731
0
                       r0 = _mm_add_epi32(r0, r2);
732
0
                       r1 = _mm_add_epi32(r1, r3);
733
0
                       r2 = _mm_add_epi32(x0, x2);
734
0
                       r3 = _mm_add_epi32(x1, x3);
735
736
0
                       r0 = _mm_add_epi32(r0, c2);
737
0
                       r1 = _mm_add_epi32(r1, c2);
738
0
                       r2 = _mm_add_epi32(r2, c2);
739
0
                       r3 = _mm_add_epi32(r3, c2);
740
741
0
                       r0 = _mm_srai_epi32(r0, shift2);
742
0
                       r1 = _mm_srai_epi32(r1, shift2);
743
0
                       r2 = _mm_srai_epi32(r2, shift2);
744
0
                       r3 = _mm_srai_epi32(r3, shift2);
745
746
0
                       r0 = _mm_packus_epi32(r0, r2);
747
0
                       r1 = _mm_packus_epi32(r1, r3);
748
0
                       r0 = _mm_packus_epi16(r0, r1);
749
750
0
                       _mm_storeu_si128((__m128i *) (dst + x), r0);
751
752
0
                   }
753
0
                   dst += dststride;
754
0
                   src1 += srcstride;
755
0
                   src2 += srcstride;
756
0
               }
757
0
    }else if(!(width & 7)){
758
0
        for (y = 0; y < height; y++) {
759
0
            for(x=0;x<width;x+=8){
760
0
                x0 = _mm_load_si128((__m128i *) (src1+x));
761
0
                x2 = _mm_load_si128((__m128i *) (src2+x));
762
763
0
                r0 = _mm_unpacklo_epi16(_mm_mullo_epi16(x0, c0),
764
0
                        _mm_mulhi_epi16(x0, c0));
765
766
0
                r2 = _mm_unpacklo_epi16(_mm_mullo_epi16(x2, c1),
767
0
                        _mm_mulhi_epi16(x2, c1));
768
769
0
                x0 = _mm_unpackhi_epi16(_mm_mullo_epi16(x0, c0),
770
0
                        _mm_mulhi_epi16(x0, c0));
771
772
0
                x2 = _mm_unpackhi_epi16(_mm_mullo_epi16(x2, c1),
773
0
                        _mm_mulhi_epi16(x2, c1));
774
775
0
                r0 = _mm_add_epi32(r0, r2);
776
0
                r2 = _mm_add_epi32(x0, x2);
777
778
779
0
                r0 = _mm_add_epi32(r0, c2);
780
0
                r2 = _mm_add_epi32(r2, c2);
781
782
0
                r0 = _mm_srai_epi32(r0, shift2);
783
0
                r2 = _mm_srai_epi32(r2, shift2);
784
785
0
                r0 = _mm_packus_epi32(r0, r2);
786
0
                r0 = _mm_packus_epi16(r0, r0);
787
788
0
                _mm_storel_epi64((__m128i *) (dst+x), r0);
789
0
            }
790
791
0
            dst += dststride;
792
0
            src1 += srcstride;
793
0
            src2 += srcstride;
794
0
        }
795
0
    }else if(!(width & 3)){
796
0
        for (y = 0; y < height; y++) {
797
0
            for(x=0;x<width;x+=4){
798
0
                x0 = _mm_loadl_epi64((__m128i *) (src1+x));
799
0
                x2 = _mm_loadl_epi64((__m128i *) (src2+x));
800
801
0
                r0 = _mm_unpacklo_epi16(_mm_mullo_epi16(x0, c0),
802
0
                        _mm_mulhi_epi16(x0, c0));
803
804
0
                r2 = _mm_unpacklo_epi16(_mm_mullo_epi16(x2, c1),
805
0
                        _mm_mulhi_epi16(x2, c1));
806
807
0
                x0 = _mm_unpackhi_epi16(_mm_mullo_epi16(x0, c0),
808
0
                        _mm_mulhi_epi16(x0, c0));
809
810
0
                x2 = _mm_unpackhi_epi16(_mm_mullo_epi16(x2, c1),
811
0
                        _mm_mulhi_epi16(x2, c1));
812
813
0
                r0 = _mm_add_epi32(r0, r2);
814
0
                r2 = _mm_add_epi32(x0, x2);
815
816
0
                r0 = _mm_add_epi32(r0, c2);
817
0
                r2 = _mm_add_epi32(r2, c2);
818
819
0
                r0 = _mm_srai_epi32(r0, shift2);
820
0
                r2 = _mm_srai_epi32(r2, shift2);
821
822
0
                r0 = _mm_packus_epi32(r0, r2);
823
0
                r0 = _mm_packus_epi16(r0, r0);
824
825
#if MASKMOVE
826
                _mm_maskmoveu_si128(r0,_mm_set_epi8(0,0,0,0,0,0,0,0,0,0,0,0,-1,-1,-1,-1),(char *) (dst+x));
827
#else
828
0
                *((uint32_t*)(dst+x)) = _mm_cvtsi128_si32(r0);
829
0
#endif
830
0
            }
831
0
            dst += dststride;
832
0
            src1 += srcstride;
833
0
            src2 += srcstride;
834
0
        }
835
0
    }else{
836
0
        for (y = 0; y < height; y++) {
837
0
            for(x=0;x<width;x+=2){
838
0
                x0 = _mm_loadl_epi64((__m128i *) (src1+x));
839
0
                x2 = _mm_loadl_epi64((__m128i *) (src2+x));
840
841
0
                r0 = _mm_unpacklo_epi16(_mm_mullo_epi16(x0, c0),
842
0
                        _mm_mulhi_epi16(x0, c0));
843
844
0
                r2 = _mm_unpacklo_epi16(_mm_mullo_epi16(x2, c1),
845
0
                        _mm_mulhi_epi16(x2, c1));
846
847
0
                x0 = _mm_unpackhi_epi16(_mm_mullo_epi16(x0, c0),
848
0
                        _mm_mulhi_epi16(x0, c0));
849
850
0
                x2 = _mm_unpackhi_epi16(_mm_mullo_epi16(x2, c1),
851
0
                        _mm_mulhi_epi16(x2, c1));
852
853
0
                r0 = _mm_add_epi32(r0, r2);
854
0
                r2 = _mm_add_epi32(x0, x2);
855
856
0
                r0 = _mm_add_epi32(r0, c2);
857
0
                r2 = _mm_add_epi32(r2, c2);
858
859
0
                r0 = _mm_srai_epi32(r0, shift2);
860
0
                r2 = _mm_srai_epi32(r2, shift2);
861
862
0
                r0 = _mm_packus_epi32(r0, r2);
863
0
                r0 = _mm_packus_epi16(r0, r0);
864
865
#if MASKMOVE
866
                _mm_maskmoveu_si128(r0,_mm_set_epi8(0,0,0,0,0,0,0,0,0,0,0,0,0,0,-1,-1),(char *) (dst+x));
867
#else
868
0
                *((uint16_t*)(dst+x)) = _mm_cvtsi128_si32(r0);
869
0
#endif
870
0
            }
871
0
            dst += dststride;
872
0
            src1 += srcstride;
873
0
            src2 += srcstride;
874
0
        }
875
0
    }
876
0
}
877
#endif
878
879
880
#if 0
881
void ff_hevc_weighted_pred_avg_sse(uint8_t denom, int16_t wl0Flag,
882
        int16_t wl1Flag, int16_t ol0Flag, int16_t ol1Flag, uint8_t *_dst,
883
                                   ptrdiff_t _dststride, const int16_t *src1, const int16_t *src2, ptrdiff_t srcstride,
884
        int width, int height) {
885
    int shift, shift2;
886
    int log2Wd;
887
    int o0;
888
    int o1;
889
    int x, y;
890
    uint8_t *dst = (uint8_t*) _dst;
891
    ptrdiff_t dststride = _dststride / sizeof(uint8_t);
892
    __m128i x0, x1, x2, x3, r0, r1, r2, r3, c0, c1, c2;
893
    shift = 14 - BIT_DEPTH;
894
    log2Wd = denom + shift;
895
896
    o0 = (ol0Flag) * (1 << (BIT_DEPTH - 8));
897
    o1 = (ol1Flag) * (1 << (BIT_DEPTH - 8));
898
    shift2 = (log2Wd + 1);
899
    c0 = _mm_set1_epi16(wl0Flag);
900
    c1 = _mm_set1_epi16(wl1Flag);
901
    c2 = _mm_set1_epi32((o0 + o1 + 1) << log2Wd);
902
903
    for (y = 0; y < height; y++) {
904
        for (x = 0; x < width; x += 16) {
905
            x0 = _mm_load_si128((__m128i *) &src1[x]);
906
            x1 = _mm_load_si128((__m128i *) &src1[x + 8]);
907
            x2 = _mm_load_si128((__m128i *) &src2[x]);
908
            x3 = _mm_load_si128((__m128i *) &src2[x + 8]);
909
910
            r0 = _mm_unpacklo_epi16(_mm_mullo_epi16(x0, c0),
911
                    _mm_mulhi_epi16(x0, c0));
912
            r1 = _mm_unpacklo_epi16(_mm_mullo_epi16(x1, c0),
913
                    _mm_mulhi_epi16(x1, c0));
914
            r2 = _mm_unpacklo_epi16(_mm_mullo_epi16(x2, c1),
915
                    _mm_mulhi_epi16(x2, c1));
916
            r3 = _mm_unpacklo_epi16(_mm_mullo_epi16(x3, c1),
917
                    _mm_mulhi_epi16(x3, c1));
918
            x0 = _mm_unpackhi_epi16(_mm_mullo_epi16(x0, c0),
919
                    _mm_mulhi_epi16(x0, c0));
920
            x1 = _mm_unpackhi_epi16(_mm_mullo_epi16(x1, c0),
921
                    _mm_mulhi_epi16(x1, c0));
922
            x2 = _mm_unpackhi_epi16(_mm_mullo_epi16(x2, c1),
923
                    _mm_mulhi_epi16(x2, c1));
924
            x3 = _mm_unpackhi_epi16(_mm_mullo_epi16(x3, c1),
925
                    _mm_mulhi_epi16(x3, c1));
926
            r0 = _mm_add_epi32(r0, r2);
927
            r1 = _mm_add_epi32(r1, r3);
928
            r2 = _mm_add_epi32(x0, x2);
929
            r3 = _mm_add_epi32(x1, x3);
930
931
            r0 = _mm_add_epi32(r0, c2);
932
            r1 = _mm_add_epi32(r1, c2);
933
            r2 = _mm_add_epi32(r2, c2);
934
            r3 = _mm_add_epi32(r3, c2);
935
936
            r0 = _mm_srai_epi32(r0, shift2);
937
            r1 = _mm_srai_epi32(r1, shift2);
938
            r2 = _mm_srai_epi32(r2, shift2);
939
            r3 = _mm_srai_epi32(r3, shift2);
940
941
            r0 = _mm_packus_epi32(r0, r2);
942
            r1 = _mm_packus_epi32(r1, r3);
943
            r0 = _mm_packus_epi16(r0, r1);
944
945
            _mm_storeu_si128((__m128i *) (dst + x), r0);
946
947
        }
948
        dst += dststride;
949
        src1 += srcstride;
950
        src2 += srcstride;
951
    }
952
}
953
#endif
954
955
956
void ff_hevc_put_hevc_epel_pixels_8_sse(int16_t *dst, ptrdiff_t dststride,
957
                                        const uint8_t *_src, ptrdiff_t srcstride,
958
                                        int width, int height, int mx,
959
6.64M
                                        int my, int16_t* mcbuffer) {
960
6.64M
    int x, y;
961
6.64M
    __m128i x1, x2,x3;
962
6.64M
    uint8_t *src = (uint8_t*) _src;
963
6.64M
    if(!(width & 15)){
964
263k
        x3= _mm_setzero_si128();
965
4.36M
        for (y = 0; y < height; y++) {
966
8.97M
                    for (x = 0; x < width; x += 16) {
967
968
4.86M
                        x1 = _mm_loadu_si128((__m128i *) &src[x]);
969
4.86M
                        x2 = _mm_unpacklo_epi8(x1, x3);
970
971
4.86M
                        x1 = _mm_unpackhi_epi8(x1, x3);
972
973
4.86M
                        x2 = _mm_slli_epi16(x2, 6);
974
4.86M
                        x1 = _mm_slli_epi16(x1, 6);
975
4.86M
                        _mm_store_si128((__m128i *) &dst[x], x2);
976
4.86M
                        _mm_store_si128((__m128i *) &dst[x + 8], x1);
977
978
4.86M
                    }
979
4.10M
                    src += srcstride;
980
4.10M
                    dst += dststride;
981
4.10M
                }
982
6.38M
    }else  if(!(width & 7)){
983
876k
        x1= _mm_setzero_si128();
984
8.67M
        for (y = 0; y < height; y++) {
985
15.7M
                    for (x = 0; x < width; x += 8) {
986
987
7.99M
                        x2 = _mm_loadl_epi64((__m128i *) &src[x]);
988
7.99M
                        x2 = _mm_unpacklo_epi8(x2, x1);
989
7.99M
                        x2 = _mm_slli_epi16(x2, 6);
990
7.99M
                        _mm_store_si128((__m128i *) &dst[x], x2);
991
992
7.99M
                    }
993
7.79M
                    src += srcstride;
994
7.79M
                    dst += dststride;
995
7.79M
                }
996
5.50M
    }else  if(!(width & 3)){
997
4.89M
        x1= _mm_setzero_si128();
998
26.1M
        for (y = 0; y < height; y++) {
999
42.9M
                    for (x = 0; x < width; x += 4) {
1000
1001
21.6M
                        x2 = _mm_loadl_epi64((__m128i *) &src[x]);
1002
21.6M
                        x2 = _mm_unpacklo_epi8(x2,x1);
1003
1004
21.6M
                        x2 = _mm_slli_epi16(x2, 6);
1005
1006
21.6M
                        _mm_storel_epi64((__m128i *) &dst[x], x2);
1007
1008
21.6M
                    }
1009
21.3M
                    src += srcstride;
1010
21.3M
                    dst += dststride;
1011
21.3M
                }
1012
4.89M
    }else{
1013
614k
        x1= _mm_setzero_si128();
1014
3.95M
        for (y = 0; y < height; y++) {
1015
7.16M
                    for (x = 0; x < width; x += 2) {
1016
1017
3.82M
                        x2 = _mm_loadl_epi64((__m128i *) &src[x]);
1018
3.82M
                        x2 = _mm_unpacklo_epi8(x2, x1);
1019
3.82M
                        x2 = _mm_slli_epi16(x2, 6);
1020
#if MASKMOVE
1021
                        _mm_maskmoveu_si128(x2,_mm_set_epi8(0,0,0,0,0,0,0,0,0,0,0,0,-1,-1,-1,-1),(char *) (dst+x));
1022
#else
1023
3.82M
                        *((uint32_t*)(dst+x)) = _mm_cvtsi128_si32(x2);
1024
3.82M
#endif
1025
3.82M
                    }
1026
3.34M
                    src += srcstride;
1027
3.34M
                    dst += dststride;
1028
3.34M
                }
1029
614k
    }
1030
1031
6.64M
}
1032
1033
#ifndef __native_client__
1034
void ff_hevc_put_hevc_epel_pixels_10_sse(int16_t *dst, ptrdiff_t dststride,
1035
                                         const uint8_t *_src, ptrdiff_t _srcstride,
1036
                                         int width, int height, int mx,
1037
0
                                         int my, int16_t* mcbuffer) {
1038
0
    int x, y;
1039
0
    __m128i x2;
1040
0
    uint16_t *src = (uint16_t*) _src;
1041
0
    ptrdiff_t srcstride = _srcstride>>1;
1042
0
    if(!(width & 7)){
1043
      //x1= _mm_setzero_si128();
1044
0
        for (y = 0; y < height; y++) {
1045
0
            for (x = 0; x < width; x += 8) {
1046
1047
0
                x2 = _mm_loadu_si128((__m128i *) &src[x]);
1048
0
                x2 = _mm_slli_epi16(x2, 4);         //shift 14 - BIT LENGTH
1049
0
                _mm_store_si128((__m128i *) &dst[x], x2);
1050
1051
0
            }
1052
0
            src += srcstride;
1053
0
            dst += dststride;
1054
0
        }
1055
0
    }else  if(!(width & 3)){
1056
      //x1= _mm_setzero_si128();
1057
0
        for (y = 0; y < height; y++) {
1058
0
            for (x = 0; x < width; x += 4) {
1059
1060
0
                x2 = _mm_loadl_epi64((__m128i *) &src[x]);
1061
0
                x2 = _mm_slli_epi16(x2, 4);     //shift 14 - BIT LENGTH
1062
1063
0
                _mm_storel_epi64((__m128i *) &dst[x], x2);
1064
1065
0
            }
1066
0
            src += srcstride;
1067
0
            dst += dststride;
1068
0
        }
1069
0
    }else{
1070
      //x1= _mm_setzero_si128();
1071
0
        for (y = 0; y < height; y++) {
1072
0
            for (x = 0; x < width; x += 2) {
1073
1074
0
                x2 = _mm_loadl_epi64((__m128i *) &src[x]);
1075
0
                x2 = _mm_slli_epi16(x2, 4);     //shift 14 - BIT LENGTH
1076
0
                _mm_maskmoveu_si128(x2,_mm_set_epi8(0,0,0,0,0,0,0,0,0,0,0,0,-1,-1,-1,-1),(char *) (dst+x));
1077
0
            }
1078
0
            src += srcstride;
1079
0
            dst += dststride;
1080
0
        }
1081
0
    }
1082
1083
0
}
1084
#endif
1085
1086
void ff_hevc_put_hevc_epel_h_8_sse(int16_t *dst, ptrdiff_t dststride,
1087
                                   const uint8_t *_src, ptrdiff_t _srcstride,
1088
                                   int width, int height, int mx,
1089
1.44M
                                   int my, int16_t* mcbuffer, int bit_depth) {
1090
1.44M
    int x, y;
1091
1.44M
    const uint8_t *src = (const uint8_t*) _src;
1092
1.44M
    ptrdiff_t srcstride = _srcstride;
1093
1.44M
    const int8_t *filter = epel_filters[mx - 1];
1094
1.44M
    __m128i r0, bshuffle1, bshuffle2, x1, x2, x3;
1095
1.44M
    int8_t filter_0 = filter[0];
1096
1.44M
    int8_t filter_1 = filter[1];
1097
1.44M
    int8_t filter_2 = filter[2];
1098
1.44M
    int8_t filter_3 = filter[3];
1099
1.44M
    r0 = _mm_set_epi8(filter_3, filter_2, filter_1, filter_0, filter_3,
1100
1.44M
            filter_2, filter_1, filter_0, filter_3, filter_2, filter_1,
1101
1.44M
            filter_0, filter_3, filter_2, filter_1, filter_0);
1102
1.44M
    bshuffle1 = _mm_set_epi8(6, 5, 4, 3, 5, 4, 3, 2, 4, 3, 2, 1, 3, 2, 1, 0);
1103
1104
1105
    /*
1106
  printf("---IN---SSE\n");
1107
1108
  int extra_top  = 1;
1109
  int extra_left = 1;
1110
  int extra_right  = 2;
1111
  int extra_bottom = 2;
1112
1113
  for (int y=-extra_top;y<height+extra_bottom;y++) {
1114
    uint8_t* p = &_src[y*_srcstride -extra_left];
1115
1116
    for (int x=-extra_left;x<width+extra_right;x++) {
1117
      printf("%05d ",*p << 6);
1118
      p++;
1119
    }
1120
    printf("\n");
1121
  }
1122
    */
1123
1124
1.44M
    if(!(width & 7)){
1125
249k
        bshuffle2 = _mm_set_epi8(10, 9, 8, 7, 9, 8, 7, 6, 8, 7, 6, 5, 7, 6, 5,
1126
249k
                        4);
1127
2.75M
                for (y = 0; y < height; y++) {
1128
6.13M
                    for (x = 0; x < width; x += 8) {
1129
1130
3.62M
                        x1 = _mm_loadu_si128((__m128i *) &src[x - 1]);
1131
3.62M
                        x2 = _mm_shuffle_epi8(x1, bshuffle1);
1132
3.62M
                        x3 = _mm_shuffle_epi8(x1, bshuffle2);
1133
1134
                        /*  PMADDUBSW then PMADDW     */
1135
3.62M
                        x2 = _mm_maddubs_epi16(x2, r0);
1136
3.62M
                        x3 = _mm_maddubs_epi16(x3, r0);
1137
3.62M
                        x2 = _mm_hadd_epi16(x2, x3);
1138
3.62M
                        _mm_store_si128((__m128i *) &dst[x], x2);
1139
3.62M
                    }
1140
2.50M
                    src += srcstride;
1141
2.50M
                    dst += dststride;
1142
2.50M
                }
1143
1.19M
    }else if(!(width & 3)){
1144
1145
5.39M
        for (y = 0; y < height; y++) {
1146
8.90M
            for (x = 0; x < width; x += 4) {
1147
            /* load data in register     */
1148
4.48M
            x1 = _mm_loadu_si128((__m128i *) &src[x-1]);
1149
4.48M
            x2 = _mm_shuffle_epi8(x1, bshuffle1);
1150
1151
            /*  PMADDUBSW then PMADDW     */
1152
4.48M
            x2 = _mm_maddubs_epi16(x2, r0);
1153
4.48M
            x2 = _mm_hadd_epi16(x2, _mm_setzero_si128());
1154
            /* give results back            */
1155
4.48M
            _mm_storel_epi64((__m128i *) &dst[x], x2);
1156
4.48M
            }
1157
4.42M
            src += srcstride;
1158
4.42M
            dst += dststride;
1159
4.42M
        }
1160
965k
    }else{
1161
1.57M
        for (y = 0; y < height; y++) {
1162
2.80M
            for (x = 0; x < width; x += 2) {
1163
            /* load data in register     */
1164
1.46M
            x1 = _mm_loadu_si128((__m128i *) &src[x-1]);
1165
1.46M
            x2 = _mm_shuffle_epi8(x1, bshuffle1);
1166
1167
            /*  PMADDUBSW then PMADDW     */
1168
1.46M
            x2 = _mm_maddubs_epi16(x2, r0);
1169
1.46M
            x2 = _mm_hadd_epi16(x2, _mm_setzero_si128());
1170
            /* give results back            */
1171
#if MASKMOVE
1172
            _mm_maskmoveu_si128(x2,_mm_set_epi8(0,0,0,0,0,0,0,0,0,0,0,0,-1,-1,-1,-1),(char *) (dst+x));
1173
#else
1174
1.46M
            *((uint32_t*)(dst+x)) = _mm_cvtsi128_si32(x2);
1175
1.46M
#endif
1176
1.46M
            }
1177
1.33M
            src += srcstride;
1178
1.33M
            dst += dststride;
1179
1.33M
        }
1180
234k
    }
1181
1.44M
}
1182
1183
#ifndef __native_client__
1184
void ff_hevc_put_hevc_epel_h_10_sse(int16_t *dst, ptrdiff_t dststride,
1185
                                    const uint8_t *_src, ptrdiff_t _srcstride,
1186
                                    int width, int height, int mx,
1187
0
                                    int my, int16_t* mcbuffer) {
1188
0
    int x, y;
1189
0
    uint16_t *src = (uint16_t*) _src;
1190
0
    ptrdiff_t srcstride = _srcstride>>1;
1191
0
    const int8_t *filter = epel_filters[mx - 1];
1192
0
    __m128i r0, bshuffle1, bshuffle2, x1, x2, x3, r1;
1193
0
    int8_t filter_0 = filter[0];
1194
0
    int8_t filter_1 = filter[1];
1195
0
    int8_t filter_2 = filter[2];
1196
0
    int8_t filter_3 = filter[3];
1197
0
    r0 = _mm_set_epi16(filter_3, filter_2, filter_1,
1198
0
            filter_0, filter_3, filter_2, filter_1, filter_0);
1199
0
    bshuffle1 = _mm_set_epi8(9,8,7,6,5,4, 3, 2,7,6,5,4, 3, 2, 1, 0);
1200
1201
0
    if(!(width & 3)){
1202
0
        bshuffle2 = _mm_set_epi8(13,12,11,10,9,8,7,6,11,10, 9,8,7,6,5, 4);
1203
0
        for (y = 0; y < height; y++) {
1204
0
            for (x = 0; x < width; x += 4) {
1205
1206
0
                x1 = _mm_loadu_si128((__m128i *) &src[x-1]);
1207
0
                x2 = _mm_shuffle_epi8(x1, bshuffle1);
1208
0
                x3 = _mm_shuffle_epi8(x1, bshuffle2);
1209
1210
1211
0
                x2 = _mm_madd_epi16(x2, r0);
1212
0
                x3 = _mm_madd_epi16(x3, r0);
1213
0
                x2 = _mm_hadd_epi32(x2, x3);
1214
0
                x2= _mm_srai_epi32(x2,2);   //>> (BIT_DEPTH - 8)
1215
1216
0
                x2 = _mm_packs_epi32(x2,r0);
1217
                //give results back
1218
0
                _mm_storel_epi64((__m128i *) &dst[x], x2);
1219
0
            }
1220
0
            src += srcstride;
1221
0
            dst += dststride;
1222
0
        }
1223
0
    }else{
1224
0
        r1= _mm_setzero_si128();
1225
0
        for (y = 0; y < height; y++) {
1226
0
            for (x = 0; x < width; x += 2) {
1227
                /* load data in register     */
1228
0
                x1 = _mm_loadu_si128((__m128i *) &src[x-1]);
1229
0
                x2 = _mm_shuffle_epi8(x1, bshuffle1);
1230
1231
                /*  PMADDUBSW then PMADDW     */
1232
0
                x2 = _mm_madd_epi16(x2, r0);
1233
0
                x2 = _mm_hadd_epi32(x2, r1);
1234
0
                x2= _mm_srai_epi32(x2,2);   //>> (BIT_DEPTH - 8)
1235
0
                x2 = _mm_packs_epi32(x2, r1);
1236
                /* give results back            */
1237
0
                _mm_maskmoveu_si128(x2,_mm_set_epi8(0,0,0,0,0,0,0,0,0,0,0,0,-1,-1,-1,-1),(char *) (dst+x));
1238
0
            }
1239
0
            src += srcstride;
1240
0
            dst += dststride;
1241
0
        }
1242
0
    }
1243
0
}
1244
#endif
1245
1246
1247
void ff_hevc_put_hevc_epel_v_8_sse(int16_t *dst, ptrdiff_t dststride,
1248
                                   const uint8_t *_src, ptrdiff_t _srcstride, int width, int height, int mx,
1249
1.56M
                                   int my, int16_t* mcbuffer, int bit_depth) {
1250
1.56M
    int x, y;
1251
1.56M
    __m128i x0, x1, x2, x3, t0, t1, t2, t3, r0, f0, f1, f2, f3, r1;
1252
1.56M
    uint8_t *src = (uint8_t*) _src;
1253
1.56M
    ptrdiff_t srcstride = _srcstride / sizeof(uint8_t);
1254
1.56M
    const int8_t *filter = epel_filters[my - 1];
1255
1.56M
    int8_t filter_0 = filter[0];
1256
1.56M
    int8_t filter_1 = filter[1];
1257
1.56M
    int8_t filter_2 = filter[2];
1258
1.56M
    int8_t filter_3 = filter[3];
1259
1.56M
    f0 = _mm_set1_epi16(filter_0);
1260
1.56M
    f1 = _mm_set1_epi16(filter_1);
1261
1.56M
    f2 = _mm_set1_epi16(filter_2);
1262
1.56M
    f3 = _mm_set1_epi16(filter_3);
1263
1264
1.56M
    if(!(width & 15)){
1265
1.04M
        for (y = 0; y < height; y++) {
1266
2.05M
            for (x = 0; x < width; x += 16) {
1267
                /* check if memory needs to be reloaded */
1268
1269
1.07M
                x0 = _mm_loadu_si128((__m128i *) &src[x - srcstride]);
1270
1.07M
                x1 = _mm_loadu_si128((__m128i *) &src[x]);
1271
1.07M
                x2 = _mm_loadu_si128((__m128i *) &src[x + srcstride]);
1272
1.07M
                x3 = _mm_loadu_si128((__m128i *) &src[x + 2 * srcstride]);
1273
1274
1.07M
                t0 = _mm_unpacklo_epi8(x0, _mm_setzero_si128());
1275
1.07M
                t1 = _mm_unpacklo_epi8(x1, _mm_setzero_si128());
1276
1.07M
                t2 = _mm_unpacklo_epi8(x2, _mm_setzero_si128());
1277
1.07M
                t3 = _mm_unpacklo_epi8(x3, _mm_setzero_si128());
1278
1279
1.07M
                x0 = _mm_unpackhi_epi8(x0, _mm_setzero_si128());
1280
1.07M
                x1 = _mm_unpackhi_epi8(x1, _mm_setzero_si128());
1281
1.07M
                x2 = _mm_unpackhi_epi8(x2, _mm_setzero_si128());
1282
1.07M
                x3 = _mm_unpackhi_epi8(x3, _mm_setzero_si128());
1283
1284
                /* multiply by correct value : */
1285
1.07M
                r0 = _mm_mullo_epi16(t0, f0);
1286
1.07M
                r1 = _mm_mullo_epi16(x0, f0);
1287
1.07M
                r0 = _mm_adds_epi16(r0, _mm_mullo_epi16(t1, f1));
1288
1.07M
                r1 = _mm_adds_epi16(r1, _mm_mullo_epi16(x1, f1));
1289
1.07M
                r0 = _mm_adds_epi16(r0, _mm_mullo_epi16(t2, f2));
1290
1.07M
                r1 = _mm_adds_epi16(r1, _mm_mullo_epi16(x2, f2));
1291
1.07M
                r0 = _mm_adds_epi16(r0, _mm_mullo_epi16(t3, f3));
1292
1.07M
                r1 = _mm_adds_epi16(r1, _mm_mullo_epi16(x3, f3));
1293
                /* give results back            */
1294
1.07M
                _mm_store_si128((__m128i *) &dst[x], r0);
1295
1.07M
                _mm_storeu_si128((__m128i *) &dst[x + 8], r1);
1296
1.07M
            }
1297
975k
            src += srcstride;
1298
975k
            dst += dststride;
1299
975k
        }
1300
1.49M
    }else if(!(width & 7)){
1301
185k
        r1= _mm_setzero_si128();
1302
1.72M
        for (y = 0; y < height; y++) {
1303
3.12M
            for(x=0;x<width;x+=8){
1304
1.58M
                x0 = _mm_loadl_epi64((__m128i *) &src[x - srcstride]);
1305
1.58M
                x1 = _mm_loadl_epi64((__m128i *) &src[x]);
1306
1.58M
                x2 = _mm_loadl_epi64((__m128i *) &src[x + srcstride]);
1307
1.58M
                x3 = _mm_loadl_epi64((__m128i *) &src[x + 2 * srcstride]);
1308
1309
1.58M
                t0 = _mm_unpacklo_epi8(x0, r1);
1310
1.58M
                t1 = _mm_unpacklo_epi8(x1, r1);
1311
1.58M
                t2 = _mm_unpacklo_epi8(x2, r1);
1312
1.58M
                t3 = _mm_unpacklo_epi8(x3, r1);
1313
1314
1315
                /* multiply by correct value : */
1316
1.58M
                r0 = _mm_mullo_epi16(t0, f0);
1317
1.58M
                r0 = _mm_adds_epi16(r0, _mm_mullo_epi16(t1, f1));
1318
1.58M
                r0 = _mm_adds_epi16(r0, _mm_mullo_epi16(t2, f2));
1319
1.58M
                r0 = _mm_adds_epi16(r0, _mm_mullo_epi16(t3, f3));
1320
                /* give results back            */
1321
1.58M
                _mm_storeu_si128((__m128i *) &dst[x], r0);
1322
1.58M
            }
1323
1.54M
            src += srcstride;
1324
1.54M
            dst += dststride;
1325
1.54M
        }
1326
1.30M
    }else if(!(width & 3)){
1327
1.14M
        r1= _mm_setzero_si128();
1328
6.00M
        for (y = 0; y < height; y++) {
1329
9.79M
            for(x=0;x<width;x+=4){
1330
4.93M
                x0 = _mm_loadl_epi64((__m128i *) &src[x - srcstride]);
1331
4.93M
                x1 = _mm_loadl_epi64((__m128i *) &src[x]);
1332
4.93M
                x2 = _mm_loadl_epi64((__m128i *) &src[x + srcstride]);
1333
4.93M
                x3 = _mm_loadl_epi64((__m128i *) &src[x + 2 * srcstride]);
1334
1335
4.93M
                t0 = _mm_unpacklo_epi8(x0, r1);
1336
4.93M
                t1 = _mm_unpacklo_epi8(x1, r1);
1337
4.93M
                t2 = _mm_unpacklo_epi8(x2, r1);
1338
4.93M
                t3 = _mm_unpacklo_epi8(x3, r1);
1339
1340
1341
                /* multiply by correct value : */
1342
4.93M
                r0 = _mm_mullo_epi16(t0, f0);
1343
4.93M
                r0 = _mm_adds_epi16(r0, _mm_mullo_epi16(t1, f1));
1344
4.93M
                r0 = _mm_adds_epi16(r0, _mm_mullo_epi16(t2, f2));
1345
4.93M
                r0 = _mm_adds_epi16(r0, _mm_mullo_epi16(t3, f3));
1346
                /* give results back            */
1347
4.93M
                _mm_storel_epi64((__m128i *) &dst[x], r0);
1348
4.93M
            }
1349
4.85M
            src += srcstride;
1350
4.85M
            dst += dststride;
1351
4.85M
        }
1352
1.14M
    }else{
1353
162k
        r1= _mm_setzero_si128();
1354
1.09M
        for (y = 0; y < height; y++) {
1355
1.93M
            for(x=0;x<width;x+=2){
1356
1.00M
                x0 = _mm_loadl_epi64((__m128i *) &src[x - srcstride]);
1357
1.00M
                x1 = _mm_loadl_epi64((__m128i *) &src[x]);
1358
1.00M
                x2 = _mm_loadl_epi64((__m128i *) &src[x + srcstride]);
1359
1.00M
                x3 = _mm_loadl_epi64((__m128i *) &src[x + 2 * srcstride]);
1360
1361
1.00M
                t0 = _mm_unpacklo_epi8(x0, r1);
1362
1.00M
                t1 = _mm_unpacklo_epi8(x1, r1);
1363
1.00M
                t2 = _mm_unpacklo_epi8(x2, r1);
1364
1.00M
                t3 = _mm_unpacklo_epi8(x3, r1);
1365
1366
1367
                /* multiply by correct value : */
1368
1.00M
                r0 = _mm_mullo_epi16(t0, f0);
1369
1.00M
                r0 = _mm_adds_epi16(r0, _mm_mullo_epi16(t1, f1));
1370
1.00M
                r0 = _mm_adds_epi16(r0, _mm_mullo_epi16(t2, f2));
1371
1.00M
                r0 = _mm_adds_epi16(r0, _mm_mullo_epi16(t3, f3));
1372
                /* give results back            */
1373
#if MASKMOVE
1374
                _mm_maskmoveu_si128(r0,_mm_set_epi8(0,0,0,0,0,0,0,0,0,0,0,0,-1,-1,-1,-1),(char *) (dst+x));
1375
#else
1376
1.00M
                *((uint32_t*)(dst+x)) = _mm_cvtsi128_si32(r0);
1377
1.00M
#endif
1378
1.00M
            }
1379
932k
            src += srcstride;
1380
932k
            dst += dststride;
1381
932k
        }
1382
162k
    }
1383
1.56M
}
1384
1385
#ifndef __native_client__
1386
void ff_hevc_put_hevc_epel_v_10_sse(int16_t *dst, ptrdiff_t dststride,
1387
                                    const uint8_t *_src, ptrdiff_t _srcstride, int width, int height, int mx,
1388
0
        int my, int16_t* mcbuffer) {
1389
0
    int x, y;
1390
0
    __m128i x0, x1, x2, x3, t0, t1, t2, t3, r0, f0, f1, f2, f3, r1, r2, r3;
1391
0
    uint16_t *src = (uint16_t*) _src;
1392
0
    ptrdiff_t srcstride = _srcstride >>1;
1393
0
    const int8_t *filter = epel_filters[my - 1];
1394
0
    int8_t filter_0 = filter[0];
1395
0
    int8_t filter_1 = filter[1];
1396
0
    int8_t filter_2 = filter[2];
1397
0
    int8_t filter_3 = filter[3];
1398
0
    f0 = _mm_set1_epi16(filter_0);
1399
0
    f1 = _mm_set1_epi16(filter_1);
1400
0
    f2 = _mm_set1_epi16(filter_2);
1401
0
    f3 = _mm_set1_epi16(filter_3);
1402
1403
0
    if(!(width & 7)){
1404
0
        r1= _mm_setzero_si128();
1405
0
        for (y = 0; y < height; y++) {
1406
0
            for(x=0;x<width;x+=8){
1407
0
                x0 = _mm_loadu_si128((__m128i *) &src[x - srcstride]);
1408
0
                x1 = _mm_loadu_si128((__m128i *) &src[x]);
1409
0
                x2 = _mm_loadu_si128((__m128i *) &src[x + srcstride]);
1410
0
                x3 = _mm_loadu_si128((__m128i *) &src[x + 2 * srcstride]);
1411
1412
                // multiply by correct value :
1413
0
                r0 = _mm_mullo_epi16(x0, f0);
1414
0
                t0 = _mm_mulhi_epi16(x0, f0);
1415
1416
0
                x0= _mm_unpacklo_epi16(r0,t0);
1417
0
                t0= _mm_unpackhi_epi16(r0,t0);
1418
1419
0
                r1 = _mm_mullo_epi16(x1, f1);
1420
0
                t1 = _mm_mulhi_epi16(x1, f1);
1421
1422
0
                x1= _mm_unpacklo_epi16(r1,t1);
1423
0
                t1= _mm_unpackhi_epi16(r1,t1);
1424
1425
1426
0
                r2 = _mm_mullo_epi16(x2, f2);
1427
0
                t2 = _mm_mulhi_epi16(x2, f2);
1428
1429
0
                x2= _mm_unpacklo_epi16(r2,t2);
1430
0
                t2= _mm_unpackhi_epi16(r2,t2);
1431
1432
1433
0
                r3 = _mm_mullo_epi16(x3, f3);
1434
0
                t3 = _mm_mulhi_epi16(x3, f3);
1435
1436
0
                x3= _mm_unpacklo_epi16(r3,t3);
1437
0
                t3= _mm_unpackhi_epi16(r3,t3);
1438
1439
1440
0
                r0= _mm_add_epi32(x0,x1);
1441
0
                r1= _mm_add_epi32(x2,x3);
1442
1443
0
                t0= _mm_add_epi32(t0,t1);
1444
0
                t1= _mm_add_epi32(t2,t3);
1445
1446
0
                r0= _mm_add_epi32(r0,r1);
1447
0
                t0= _mm_add_epi32(t0,t1);
1448
1449
0
                r0= _mm_srai_epi32(r0,2);//>> (BIT_DEPTH - 8)
1450
0
                t0= _mm_srai_epi32(t0,2);//>> (BIT_DEPTH - 8)
1451
1452
0
                r0= _mm_packs_epi32(r0, t0);
1453
                // give results back
1454
0
                _mm_storeu_si128((__m128i *) &dst[x], r0);
1455
0
            }
1456
0
            src += srcstride;
1457
0
            dst += dststride;
1458
0
        }
1459
0
    }else if(!(width & 3)){
1460
0
        r1= _mm_setzero_si128();
1461
0
        for (y = 0; y < height; y++) {
1462
0
            for(x=0;x<width;x+=4){
1463
0
                x0 = _mm_loadl_epi64((__m128i *) &src[x - srcstride]);
1464
0
                x1 = _mm_loadl_epi64((__m128i *) &src[x]);
1465
0
                x2 = _mm_loadl_epi64((__m128i *) &src[x + srcstride]);
1466
0
                x3 = _mm_loadl_epi64((__m128i *) &src[x + 2 * srcstride]);
1467
1468
                /* multiply by correct value : */
1469
0
                r0 = _mm_mullo_epi16(x0, f0);
1470
0
                t0 = _mm_mulhi_epi16(x0, f0);
1471
1472
0
                x0= _mm_unpacklo_epi16(r0,t0);
1473
1474
0
                r1 = _mm_mullo_epi16(x1, f1);
1475
0
                t1 = _mm_mulhi_epi16(x1, f1);
1476
1477
0
                x1= _mm_unpacklo_epi16(r1,t1);
1478
1479
1480
0
                r2 = _mm_mullo_epi16(x2, f2);
1481
0
                t2 = _mm_mulhi_epi16(x2, f2);
1482
1483
0
                x2= _mm_unpacklo_epi16(r2,t2);
1484
1485
1486
0
                r3 = _mm_mullo_epi16(x3, f3);
1487
0
                t3 = _mm_mulhi_epi16(x3, f3);
1488
1489
0
                x3= _mm_unpacklo_epi16(r3,t3);
1490
1491
1492
0
                r0= _mm_add_epi32(x0,x1);
1493
0
                r1= _mm_add_epi32(x2,x3);
1494
0
                r0= _mm_add_epi32(r0,r1);
1495
0
                r0= _mm_srai_epi32(r0,2);//>> (BIT_DEPTH - 8)
1496
1497
0
                r0= _mm_packs_epi32(r0, r0);
1498
1499
                // give results back
1500
0
                _mm_storel_epi64((__m128i *) &dst[x], r0);
1501
0
            }
1502
0
            src += srcstride;
1503
0
            dst += dststride;
1504
0
        }
1505
0
    }else{
1506
0
        r1= _mm_setzero_si128();
1507
0
        for (y = 0; y < height; y++) {
1508
0
            for(x=0;x<width;x+=2){
1509
0
                x0 = _mm_loadl_epi64((__m128i *) &src[x - srcstride]);
1510
0
                x1 = _mm_loadl_epi64((__m128i *) &src[x]);
1511
0
                x2 = _mm_loadl_epi64((__m128i *) &src[x + srcstride]);
1512
0
                x3 = _mm_loadl_epi64((__m128i *) &src[x + 2 * srcstride]);
1513
1514
                /* multiply by correct value : */
1515
0
                r0 = _mm_mullo_epi16(x0, f0);
1516
0
                t0 = _mm_mulhi_epi16(x0, f0);
1517
1518
0
                x0= _mm_unpacklo_epi16(r0,t0);
1519
1520
0
                r1 = _mm_mullo_epi16(x1, f1);
1521
0
                t1 = _mm_mulhi_epi16(x1, f1);
1522
1523
0
                x1= _mm_unpacklo_epi16(r1,t1);
1524
1525
0
                r2 = _mm_mullo_epi16(x2, f2);
1526
0
                t2 = _mm_mulhi_epi16(x2, f2);
1527
1528
0
                x2= _mm_unpacklo_epi16(r2,t2);
1529
1530
0
                r3 = _mm_mullo_epi16(x3, f3);
1531
0
                t3 = _mm_mulhi_epi16(x3, f3);
1532
1533
0
                x3= _mm_unpacklo_epi16(r3,t3);
1534
1535
0
                r0= _mm_add_epi32(x0,x1);
1536
0
                r1= _mm_add_epi32(x2,x3);
1537
0
                r0= _mm_add_epi32(r0,r1);
1538
0
                r0= _mm_srai_epi32(r0,2);//>> (BIT_DEPTH - 8)
1539
1540
0
                r0= _mm_packs_epi32(r0, r0);
1541
1542
                /* give results back            */
1543
0
                _mm_maskmoveu_si128(r0,_mm_set_epi8(0,0,0,0,0,0,0,0,0,0,0,0,-1,-1,-1,-1),(char *) (dst+x));
1544
1545
0
            }
1546
0
            src += srcstride;
1547
0
            dst += dststride;
1548
0
        }
1549
0
    }
1550
0
}
1551
#endif
1552
1553
void ff_hevc_put_hevc_epel_hv_8_sse(int16_t *dst, ptrdiff_t dststride,
1554
                                    const uint8_t *_src, ptrdiff_t _srcstride, int width, int height, int mx,
1555
4.69M
                                    int my, int16_t* mcbuffer, int bit_depth) {
1556
4.69M
  int x, y;
1557
4.69M
  uint8_t *src = (uint8_t*) _src;
1558
4.69M
  ptrdiff_t srcstride = _srcstride;
1559
4.69M
  const int8_t *filter_h = epel_filters[mx - 1];
1560
4.69M
  const int8_t *filter_v = epel_filters[my - 1];
1561
4.69M
  __m128i r0, bshuffle1, bshuffle2, x0, x1, x2, x3, t0, t1, t2, t3, f0, f1,
1562
4.69M
  f2, f3, r1, r2;
1563
4.69M
  int8_t filter_0 = filter_h[0];
1564
4.69M
  int8_t filter_1 = filter_h[1];
1565
4.69M
  int8_t filter_2 = filter_h[2];
1566
4.69M
  int8_t filter_3 = filter_h[3];
1567
4.69M
  int16_t *tmp = mcbuffer;
1568
4.69M
  r0 = _mm_set_epi8(filter_3, filter_2, filter_1, filter_0, filter_3,
1569
4.69M
      filter_2, filter_1, filter_0, filter_3, filter_2, filter_1,
1570
4.69M
      filter_0, filter_3, filter_2, filter_1, filter_0);
1571
4.69M
  bshuffle1 = _mm_set_epi8(6, 5, 4, 3, 5, 4, 3, 2, 4, 3, 2, 1, 3, 2, 1, 0);
1572
1573
4.69M
  src -= epel_extra_before * srcstride;
1574
1575
4.69M
  f3 = _mm_set1_epi16(filter_v[3]);
1576
4.69M
  f1 = _mm_set1_epi16(filter_v[1]);
1577
4.69M
  f2 = _mm_set1_epi16(filter_v[2]);
1578
4.69M
  f0 = _mm_set1_epi16(filter_v[0]);
1579
1580
  /* horizontal treatment */
1581
4.69M
  if(!(width & 7)){
1582
712k
    bshuffle2 = _mm_set_epi8(10, 9, 8, 7, 9, 8, 7, 6, 8, 7, 6, 5, 7, 6, 5,
1583
712k
        4);
1584
9.71M
    for (y = 0; y < height + epel_extra; y++) {
1585
21.7M
      for (x = 0; x < width; x += 8) {
1586
1587
12.7M
        x1 = _mm_loadu_si128((__m128i *) &src[x - 1]);
1588
12.7M
        x2 = _mm_shuffle_epi8(x1, bshuffle1);
1589
12.7M
        x3 = _mm_shuffle_epi8(x1, bshuffle2);
1590
1591
        /*  PMADDUBSW then PMADDW     */
1592
12.7M
        x2 = _mm_maddubs_epi16(x2, r0);
1593
12.7M
        x3 = _mm_maddubs_epi16(x3, r0);
1594
12.7M
        x2 = _mm_hadd_epi16(x2, x3);
1595
12.7M
        _mm_store_si128((__m128i *) &tmp[x], x2);
1596
12.7M
      }
1597
9.00M
      src += srcstride;
1598
9.00M
      tmp += MAX_PB_SIZE;
1599
9.00M
    }
1600
712k
    tmp = mcbuffer + epel_extra_before * MAX_PB_SIZE;
1601
1602
    /* vertical treatment */
1603
1604
7.57M
    for (y = 0; y < height; y++) {
1605
16.8M
      for (x = 0; x < width; x += 8) {
1606
        /* check if memory needs to be reloaded */
1607
10.0M
        x0 = _mm_load_si128((__m128i *) &tmp[x - MAX_PB_SIZE]);
1608
10.0M
        x1 = _mm_load_si128((__m128i *) &tmp[x]);
1609
10.0M
        x2 = _mm_load_si128((__m128i *) &tmp[x + MAX_PB_SIZE]);
1610
10.0M
        x3 = _mm_load_si128((__m128i *) &tmp[x + 2 * MAX_PB_SIZE]);
1611
1612
10.0M
        r0 = _mm_mullo_epi16(x0, f0);
1613
10.0M
        r1 = _mm_mulhi_epi16(x0, f0);
1614
10.0M
        r2 = _mm_mullo_epi16(x1, f1);
1615
10.0M
        t0 = _mm_unpacklo_epi16(r0, r1);
1616
10.0M
        x0 = _mm_unpackhi_epi16(r0, r1);
1617
10.0M
        r0 = _mm_mulhi_epi16(x1, f1);
1618
10.0M
        r1 = _mm_mullo_epi16(x2, f2);
1619
10.0M
        t1 = _mm_unpacklo_epi16(r2, r0);
1620
10.0M
        x1 = _mm_unpackhi_epi16(r2, r0);
1621
10.0M
        r2 = _mm_mulhi_epi16(x2, f2);
1622
10.0M
        r0 = _mm_mullo_epi16(x3, f3);
1623
10.0M
        t2 = _mm_unpacklo_epi16(r1, r2);
1624
10.0M
        x2 = _mm_unpackhi_epi16(r1, r2);
1625
10.0M
        r1 = _mm_mulhi_epi16(x3, f3);
1626
10.0M
        t3 = _mm_unpacklo_epi16(r0, r1);
1627
10.0M
        x3 = _mm_unpackhi_epi16(r0, r1);
1628
1629
        /* multiply by correct value : */
1630
10.0M
        r0 = _mm_add_epi32(t0, t1);
1631
10.0M
        r1 = _mm_add_epi32(x0, x1);
1632
10.0M
        r0 = _mm_add_epi32(r0, t2);
1633
10.0M
        r1 = _mm_add_epi32(r1, x2);
1634
10.0M
        r0 = _mm_add_epi32(r0, t3);
1635
10.0M
        r1 = _mm_add_epi32(r1, x3);
1636
10.0M
        r0 = _mm_srai_epi32(r0, 6);
1637
10.0M
        r1 = _mm_srai_epi32(r1, 6);
1638
1639
        /* give results back            */
1640
10.0M
        r0 = _mm_packs_epi32(r0, r1);
1641
10.0M
        _mm_store_si128((__m128i *) &dst[x], r0);
1642
10.0M
      }
1643
6.86M
      tmp += MAX_PB_SIZE;
1644
6.86M
      dst += dststride;
1645
6.86M
    }
1646
3.98M
  }else if(!(width & 3)){
1647
26.4M
    for (y = 0; y < height + epel_extra; y ++) {
1648
47.0M
      for(x=0;x<width;x+=4){
1649
        /* load data in register     */
1650
23.7M
        x1 = _mm_loadl_epi64((__m128i *) &src[x-1]);
1651
1652
23.7M
        x1 = _mm_shuffle_epi8(x1, bshuffle1);
1653
1654
        /*  PMADDUBSW then PMADDW     */
1655
23.7M
        x1 = _mm_maddubs_epi16(x1, r0);
1656
23.7M
        x1 = _mm_hadd_epi16(x1, _mm_setzero_si128());
1657
1658
        /* give results back            */
1659
23.7M
        _mm_storel_epi64((__m128i *) &tmp[x], x1);
1660
1661
23.7M
      }
1662
23.3M
      src += srcstride;
1663
23.3M
      tmp += MAX_PB_SIZE;
1664
23.3M
    }
1665
3.10M
    tmp = mcbuffer + epel_extra_before * MAX_PB_SIZE;
1666
1667
    /* vertical treatment */
1668
1669
1670
17.1M
    for (y = 0; y < height; y++) {
1671
28.3M
      for (x = 0; x < width; x += 4) {
1672
        /* check if memory needs to be reloaded */
1673
14.3M
        x0 = _mm_loadl_epi64((__m128i *) &tmp[x - MAX_PB_SIZE]);
1674
14.3M
        x1 = _mm_loadl_epi64((__m128i *) &tmp[x]);
1675
14.3M
        x2 = _mm_loadl_epi64((__m128i *) &tmp[x + MAX_PB_SIZE]);
1676
14.3M
        x3 = _mm_loadl_epi64((__m128i *) &tmp[x + 2 * MAX_PB_SIZE]);
1677
1678
14.3M
        r0 = _mm_mullo_epi16(x0, f0);
1679
14.3M
        r1 = _mm_mulhi_epi16(x0, f0);
1680
14.3M
        r2 = _mm_mullo_epi16(x1, f1);
1681
14.3M
        t0 = _mm_unpacklo_epi16(r0, r1);
1682
1683
14.3M
        r0 = _mm_mulhi_epi16(x1, f1);
1684
14.3M
        r1 = _mm_mullo_epi16(x2, f2);
1685
14.3M
        t1 = _mm_unpacklo_epi16(r2, r0);
1686
1687
14.3M
        r2 = _mm_mulhi_epi16(x2, f2);
1688
14.3M
        r0 = _mm_mullo_epi16(x3, f3);
1689
14.3M
        t2 = _mm_unpacklo_epi16(r1, r2);
1690
1691
14.3M
        r1 = _mm_mulhi_epi16(x3, f3);
1692
14.3M
        t3 = _mm_unpacklo_epi16(r0, r1);
1693
1694
1695
        /* multiply by correct value : */
1696
14.3M
        r0 = _mm_add_epi32(t0, t1);
1697
14.3M
        r0 = _mm_add_epi32(r0, t2);
1698
14.3M
        r0 = _mm_add_epi32(r0, t3);
1699
14.3M
        r0 = _mm_srai_epi32(r0, 6);
1700
1701
        /* give results back            */
1702
14.3M
        r0 = _mm_packs_epi32(r0, r0);
1703
14.3M
        _mm_storel_epi64((__m128i *) &dst[x], r0);
1704
14.3M
      }
1705
14.0M
      tmp += MAX_PB_SIZE;
1706
14.0M
      dst += dststride;
1707
14.0M
    }
1708
3.10M
  }else{
1709
#if MASKMOVE
1710
    bshuffle2=_mm_set_epi8(0,0,0,0,0,0,0,0,0,0,0,0,-1,-1,-1,-1);
1711
#endif
1712
8.18M
    for (y = 0; y < height + epel_extra; y ++) {
1713
15.0M
      for(x=0;x<width;x+=2){
1714
        /* load data in register     */
1715
7.73M
        x1 = _mm_loadl_epi64((__m128i *) &src[x-1]);
1716
7.73M
        x1 = _mm_shuffle_epi8(x1, bshuffle1);
1717
1718
        /*  PMADDUBSW then PMADDW     */
1719
7.73M
        x1 = _mm_maddubs_epi16(x1, r0);
1720
7.73M
        x1 = _mm_hadd_epi16(x1, _mm_setzero_si128());
1721
1722
        /* give results back            */
1723
#if MASKMOVE
1724
        _mm_maskmoveu_si128(x1,bshuffle2,(char *) (tmp+x));
1725
#else
1726
7.73M
                                *((uint32_t*)(tmp+x)) = _mm_cvtsi128_si32(x1);
1727
7.73M
#endif
1728
7.73M
      }
1729
7.31M
      src += srcstride;
1730
7.31M
      tmp += MAX_PB_SIZE;
1731
7.31M
    }
1732
1733
871k
    tmp = mcbuffer + epel_extra_before * MAX_PB_SIZE;
1734
1735
    /* vertical treatment */
1736
1737
5.56M
    for (y = 0; y < height; y++) {
1738
9.71M
      for (x = 0; x < width; x += 2) {
1739
        /* check if memory needs to be reloaded */
1740
5.01M
        x0 = _mm_loadl_epi64((__m128i *) &tmp[x - MAX_PB_SIZE]);
1741
5.01M
        x1 = _mm_loadl_epi64((__m128i *) &tmp[x]);
1742
5.01M
        x2 = _mm_loadl_epi64((__m128i *) &tmp[x + MAX_PB_SIZE]);
1743
5.01M
        x3 = _mm_loadl_epi64((__m128i *) &tmp[x + 2 * MAX_PB_SIZE]);
1744
1745
5.01M
        r0 = _mm_mullo_epi16(x0, f0);
1746
5.01M
        r1 = _mm_mulhi_epi16(x0, f0);
1747
5.01M
        r2 = _mm_mullo_epi16(x1, f1);
1748
5.01M
        t0 = _mm_unpacklo_epi16(r0, r1);
1749
5.01M
        r0 = _mm_mulhi_epi16(x1, f1);
1750
5.01M
        r1 = _mm_mullo_epi16(x2, f2);
1751
5.01M
        t1 = _mm_unpacklo_epi16(r2, r0);
1752
5.01M
        r2 = _mm_mulhi_epi16(x2, f2);
1753
5.01M
        r0 = _mm_mullo_epi16(x3, f3);
1754
5.01M
        t2 = _mm_unpacklo_epi16(r1, r2);
1755
5.01M
        r1 = _mm_mulhi_epi16(x3, f3);
1756
5.01M
        t3 = _mm_unpacklo_epi16(r0, r1);
1757
1758
        /* multiply by correct value : */
1759
5.01M
        r0 = _mm_add_epi32(t0, t1);
1760
5.01M
        r0 = _mm_add_epi32(r0, t2);
1761
5.01M
        r0 = _mm_add_epi32(r0, t3);
1762
5.01M
        r0 = _mm_srai_epi32(r0, 6);
1763
        /* give results back            */
1764
5.01M
        r0 = _mm_packs_epi32(r0, r0);
1765
#if MASKMOVE
1766
        _mm_maskmoveu_si128(r0,bshuffle2,(char *) (dst+x));
1767
#else
1768
5.01M
                                *((uint32_t*)(dst+x)) = _mm_cvtsi128_si32(r0);
1769
5.01M
#endif
1770
5.01M
      }
1771
4.69M
      tmp += MAX_PB_SIZE;
1772
4.69M
      dst += dststride;
1773
4.69M
    }
1774
871k
  }
1775
1776
4.69M
}
1777
1778
1779
#ifndef __native_client__
1780
void ff_hevc_put_hevc_epel_hv_10_sse(int16_t *dst, ptrdiff_t dststride,
1781
                                     const uint8_t *_src, ptrdiff_t _srcstride, int width, int height, int mx,
1782
0
        int my, int16_t* mcbuffer) {
1783
0
    int x, y;
1784
0
    uint16_t *src = (uint16_t*) _src;
1785
0
    ptrdiff_t srcstride = _srcstride>>1;
1786
0
    const int8_t *filter_h = epel_filters[mx - 1];
1787
0
    const int8_t *filter_v = epel_filters[my - 1];
1788
0
    __m128i r0, bshuffle1, bshuffle2, x0, x1, x2, x3, t0, t1, t2, t3, f0, f1,
1789
0
    f2, f3, r1, r2, r3;
1790
0
    int8_t filter_0 = filter_h[0];
1791
0
    int8_t filter_1 = filter_h[1];
1792
0
    int8_t filter_2 = filter_h[2];
1793
0
    int8_t filter_3 = filter_h[3];
1794
0
    int16_t *tmp = mcbuffer;
1795
1796
0
    r0 = _mm_set_epi16(filter_3, filter_2, filter_1,
1797
0
                filter_0, filter_3, filter_2, filter_1, filter_0);
1798
0
        bshuffle1 = _mm_set_epi8(9,8,7,6,5,4, 3, 2,7,6,5,4, 3, 2, 1, 0);
1799
1800
0
    src -= epel_extra_before * srcstride;
1801
1802
0
    f0 = _mm_set1_epi16(filter_v[0]);
1803
0
    f1 = _mm_set1_epi16(filter_v[1]);
1804
0
    f2 = _mm_set1_epi16(filter_v[2]);
1805
0
    f3 = _mm_set1_epi16(filter_v[3]);
1806
1807
1808
    /* horizontal treatment */
1809
0
    if(!(width & 3)){
1810
0
        bshuffle2 = _mm_set_epi8(13,12,11,10,9,8,7,6,11,10, 9,8,7,6,5, 4);
1811
0
        for (y = 0; y < height + epel_extra; y ++) {
1812
0
            for(x=0;x<width;x+=4){
1813
1814
0
                x1 = _mm_loadu_si128((__m128i *) &src[x-1]);
1815
0
                x2 = _mm_shuffle_epi8(x1, bshuffle1);
1816
0
                x3 = _mm_shuffle_epi8(x1, bshuffle2);
1817
1818
1819
0
                x2 = _mm_madd_epi16(x2, r0);
1820
0
                x3 = _mm_madd_epi16(x3, r0);
1821
0
                x2 = _mm_hadd_epi32(x2, x3);
1822
0
                x2= _mm_srai_epi32(x2,2);   //>> (BIT_DEPTH - 8)
1823
1824
0
                x2 = _mm_packs_epi32(x2,r0);
1825
                //give results back
1826
0
                _mm_storel_epi64((__m128i *) &tmp[x], x2);
1827
1828
0
            }
1829
0
            src += srcstride;
1830
0
            tmp += MAX_PB_SIZE;
1831
0
        }
1832
0
        tmp = mcbuffer + epel_extra_before * MAX_PB_SIZE;
1833
1834
        // vertical treatment
1835
1836
1837
0
        for (y = 0; y < height; y++) {
1838
0
            for (x = 0; x < width; x += 4) {
1839
0
                x0 = _mm_loadl_epi64((__m128i *) &tmp[x - MAX_PB_SIZE]);
1840
0
                x1 = _mm_loadl_epi64((__m128i *) &tmp[x]);
1841
0
                x2 = _mm_loadl_epi64((__m128i *) &tmp[x + MAX_PB_SIZE]);
1842
0
                x3 = _mm_loadl_epi64((__m128i *) &tmp[x + 2 * MAX_PB_SIZE]);
1843
1844
0
                r0 = _mm_mullo_epi16(x0, f0);
1845
0
                r1 = _mm_mulhi_epi16(x0, f0);
1846
0
                r2 = _mm_mullo_epi16(x1, f1);
1847
0
                t0 = _mm_unpacklo_epi16(r0, r1);
1848
1849
0
                r0 = _mm_mulhi_epi16(x1, f1);
1850
0
                r1 = _mm_mullo_epi16(x2, f2);
1851
0
                t1 = _mm_unpacklo_epi16(r2, r0);
1852
1853
0
                r2 = _mm_mulhi_epi16(x2, f2);
1854
0
                r0 = _mm_mullo_epi16(x3, f3);
1855
0
                t2 = _mm_unpacklo_epi16(r1, r2);
1856
1857
0
                r1 = _mm_mulhi_epi16(x3, f3);
1858
0
                t3 = _mm_unpacklo_epi16(r0, r1);
1859
1860
1861
1862
0
                r0 = _mm_add_epi32(t0, t1);
1863
0
                r0 = _mm_add_epi32(r0, t2);
1864
0
                r0 = _mm_add_epi32(r0, t3);
1865
0
                r0 = _mm_srai_epi32(r0, 6);
1866
1867
                // give results back
1868
0
                r0 = _mm_packs_epi32(r0, r0);
1869
0
                _mm_storel_epi64((__m128i *) &dst[x], r0);
1870
0
            }
1871
0
            tmp += MAX_PB_SIZE;
1872
0
            dst += dststride;
1873
0
        }
1874
0
    }else{
1875
0
        bshuffle2=_mm_set_epi8(0,0,0,0,0,0,0,0,0,0,0,0,-1,-1,-1,-1);
1876
0
        r1= _mm_setzero_si128();
1877
0
        for (y = 0; y < height + epel_extra; y ++) {
1878
0
            for(x=0;x<width;x+=2){
1879
                /* load data in register     */
1880
0
                x1 = _mm_loadu_si128((__m128i *) &src[x-1]);
1881
0
                x2 = _mm_shuffle_epi8(x1, bshuffle1);
1882
1883
                /*  PMADDUBSW then PMADDW     */
1884
0
                x2 = _mm_madd_epi16(x2, r0);
1885
0
                x2 = _mm_hadd_epi32(x2, r1);
1886
0
                x2= _mm_srai_epi32(x2,2);   //>> (BIT_DEPTH - 8)
1887
0
                x2 = _mm_packs_epi32(x2, r1);
1888
                /* give results back            */
1889
0
                _mm_maskmoveu_si128(x2,bshuffle2,(char *) (tmp+x));
1890
0
            }
1891
0
            src += srcstride;
1892
0
            tmp += MAX_PB_SIZE;
1893
0
        }
1894
1895
0
        tmp = mcbuffer + epel_extra_before * MAX_PB_SIZE;
1896
1897
        /* vertical treatment */
1898
1899
0
        for (y = 0; y < height; y++) {
1900
0
            for (x = 0; x < width; x += 2) {
1901
                /* check if memory needs to be reloaded */
1902
0
                x0 = _mm_loadl_epi64((__m128i *) &tmp[x - MAX_PB_SIZE]);
1903
0
                x1 = _mm_loadl_epi64((__m128i *) &tmp[x]);
1904
0
                x2 = _mm_loadl_epi64((__m128i *) &tmp[x + MAX_PB_SIZE]);
1905
0
                x3 = _mm_loadl_epi64((__m128i *) &tmp[x + 2 * MAX_PB_SIZE]);
1906
1907
0
                r0 = _mm_mullo_epi16(x0, f0);
1908
0
                t0 = _mm_mulhi_epi16(x0, f0);
1909
1910
0
                x0= _mm_unpacklo_epi16(r0,t0);
1911
1912
0
                r1 = _mm_mullo_epi16(x1, f1);
1913
0
                t1 = _mm_mulhi_epi16(x1, f1);
1914
1915
0
                x1= _mm_unpacklo_epi16(r1,t1);
1916
1917
0
                r2 = _mm_mullo_epi16(x2, f2);
1918
0
                t2 = _mm_mulhi_epi16(x2, f2);
1919
1920
0
                x2= _mm_unpacklo_epi16(r2,t2);
1921
1922
0
                r3 = _mm_mullo_epi16(x3, f3);
1923
0
                t3 = _mm_mulhi_epi16(x3, f3);
1924
1925
0
                x3= _mm_unpacklo_epi16(r3,t3);
1926
1927
0
                r0= _mm_add_epi32(x0,x1);
1928
0
                r1= _mm_add_epi32(x2,x3);
1929
0
                r0= _mm_add_epi32(r0,r1);
1930
0
                r0 = _mm_srai_epi32(r0, 6);
1931
                /* give results back            */
1932
0
                r0 = _mm_packs_epi32(r0, r0);
1933
0
                _mm_maskmoveu_si128(r0,bshuffle2,(char *) (dst+x));
1934
0
            }
1935
0
            tmp += MAX_PB_SIZE;
1936
0
            dst += dststride;
1937
0
        }
1938
0
    }
1939
0
}
1940
#endif
1941
1942
void ff_hevc_put_hevc_qpel_pixels_8_sse(int16_t *dst, ptrdiff_t dststride,
1943
                                        const uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
1944
3.45M
        int16_t* mcbuffer) {
1945
3.45M
    int x, y;
1946
3.45M
    __m128i x1, x2, x3, x0;
1947
3.45M
    uint8_t *src = (uint8_t*) _src;
1948
3.45M
    ptrdiff_t srcstride = _srcstride;
1949
3.45M
    x0= _mm_setzero_si128();
1950
3.45M
    if(!(width & 15)){
1951
11.3M
        for (y = 0; y < height; y++) {
1952
26.7M
            for (x = 0; x < width; x += 16) {
1953
1954
15.9M
                x1 = _mm_loadu_si128((__m128i *) &src[x]);
1955
15.9M
                x2 = _mm_unpacklo_epi8(x1, x0);
1956
1957
15.9M
                x3 = _mm_unpackhi_epi8(x1, x0);
1958
1959
15.9M
                x2 = _mm_slli_epi16(x2, 6);
1960
15.9M
                x3 = _mm_slli_epi16(x3, 6);
1961
15.9M
                _mm_storeu_si128((__m128i *) &dst[x], x2);
1962
15.9M
                _mm_storeu_si128((__m128i *) &dst[x + 8], x3);
1963
1964
15.9M
            }
1965
10.7M
            src += srcstride;
1966
10.7M
            dst += dststride;
1967
10.7M
        }
1968
2.90M
    }else if(!(width & 7)){
1969
22.6M
        for (y = 0; y < height; y++) {
1970
40.6M
            for (x = 0; x < width; x += 8) {
1971
1972
20.4M
                x1 = _mm_loadu_si128((__m128i *) &src[x]);
1973
20.4M
                x2 = _mm_unpacklo_epi8(x1, x0);
1974
20.4M
                x2 = _mm_slli_epi16(x2, 6);
1975
20.4M
                _mm_storeu_si128((__m128i *) &dst[x], x2);
1976
1977
20.4M
            }
1978
20.1M
            src += srcstride;
1979
20.1M
            dst += dststride;
1980
20.1M
        }
1981
2.52M
    }else if(!(width & 3)){
1982
3.62M
        for (y = 0; y < height; y++) {
1983
6.90M
            for(x=0;x<width;x+=4){
1984
3.66M
                x1 = _mm_loadu_si128((__m128i *) &src[x]);
1985
3.66M
                x2 = _mm_unpacklo_epi8(x1, x0);
1986
3.66M
                x2 = _mm_slli_epi16(x2, 6);
1987
3.66M
                _mm_storel_epi64((__m128i *) &dst[x], x2);
1988
3.66M
            }
1989
3.24M
            src += srcstride;
1990
3.24M
            dst += dststride;
1991
3.24M
        }
1992
379k
    }else{
1993
#if MASKMOVE
1994
        x4= _mm_set_epi32(0,0,0,-1); //mask to store
1995
#endif
1996
0
        for (y = 0; y < height; y++) {
1997
0
                    for(x=0;x<width;x+=2){
1998
0
                        x1 = _mm_loadl_epi64((__m128i *) &src[x]);
1999
0
                        x2 = _mm_unpacklo_epi8(x1, x0);
2000
0
                        x2 = _mm_slli_epi16(x2, 6);
2001
#if MASKMOVE
2002
                        _mm_maskmoveu_si128(x2,x4,(char *) (dst+x));
2003
#else
2004
0
                        *((uint16_t*)(dst+x)) = _mm_cvtsi128_si32(x2);
2005
0
#endif
2006
0
                    }
2007
0
                    src += srcstride;
2008
0
                    dst += dststride;
2009
0
                }
2010
0
    }
2011
2012
2013
3.45M
}
2014
2015
#ifndef __native_client__
2016
void ff_hevc_put_hevc_qpel_pixels_10_sse(int16_t *dst, ptrdiff_t dststride,
2017
                                         const uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
2018
0
        int16_t* mcbuffer) {
2019
0
    int x, y;
2020
0
    __m128i x1, x2, x4;
2021
0
    uint16_t *src = (uint16_t*) _src;
2022
0
    ptrdiff_t srcstride = _srcstride>>1;
2023
0
    if(!(width & 7)){
2024
0
        for (y = 0; y < height; y++) {
2025
0
            for (x = 0; x < width; x += 8) {
2026
2027
0
                x1 = _mm_loadu_si128((__m128i *) &src[x]);
2028
0
                x2 = _mm_slli_epi16(x1, 4); //14-BIT DEPTH
2029
0
                _mm_storeu_si128((__m128i *) &dst[x], x2);
2030
2031
0
            }
2032
0
            src += srcstride;
2033
0
            dst += dststride;
2034
0
        }
2035
0
    }else if(!(width & 3)){
2036
0
        for (y = 0; y < height; y++) {
2037
0
            for(x=0;x<width;x+=4){
2038
0
                x1 = _mm_loadl_epi64((__m128i *) &src[x]);
2039
0
                x2 = _mm_slli_epi16(x1, 4);//14-BIT DEPTH
2040
0
                _mm_storel_epi64((__m128i *) &dst[x], x2);
2041
0
            }
2042
0
            src += srcstride;
2043
0
            dst += dststride;
2044
0
        }
2045
0
    }else{
2046
0
        x4= _mm_set_epi32(0,0,0,-1); //mask to store
2047
0
        for (y = 0; y < height; y++) {
2048
0
                    for(x=0;x<width;x+=2){
2049
0
                        x1 = _mm_loadl_epi64((__m128i *) &src[x]);
2050
0
                        x2 = _mm_slli_epi16(x1, 4);//14-BIT DEPTH
2051
0
                        _mm_maskmoveu_si128(x2,x4,(char *) (dst+x));
2052
0
                    }
2053
0
                    src += srcstride;
2054
0
                    dst += dststride;
2055
0
                }
2056
0
    }
2057
2058
2059
0
}
2060
#endif
2061
2062
2063
void ff_hevc_put_hevc_qpel_h_1_8_sse(int16_t *dst, ptrdiff_t dststride,
2064
                                     const uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
2065
253k
        int16_t* mcbuffer) {
2066
253k
    int x, y;
2067
253k
    const uint8_t *src = _src;
2068
253k
    ptrdiff_t srcstride = _srcstride / sizeof(uint8_t);
2069
253k
    __m128i x1, r0, x2, x3, x4, x5;
2070
2071
253k
    r0 = _mm_set_epi8(0, 1, -5, 17, 58, -10, 4, -1, 0, 1, -5, 17, 58, -10, 4,
2072
253k
            -1);
2073
2074
253k
    if(!(width & 7)){
2075
2.40M
        for (y = 0; y < height; y++) {
2076
5.70M
            for (x = 0; x < width; x += 8) {
2077
                /* load data in register     */
2078
3.51M
                x1 = _mm_loadu_si128((__m128i *) &src[x - 3]);
2079
3.51M
                x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
2080
3.51M
                x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
2081
3.51M
                        _mm_srli_si128(x1, 3));
2082
3.51M
                x4 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 4),
2083
3.51M
                        _mm_srli_si128(x1, 5));
2084
3.51M
                x5 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 6),
2085
3.51M
                        _mm_srli_si128(x1, 7));
2086
2087
                /*  PMADDUBSW then PMADDW     */
2088
3.51M
                x2 = _mm_maddubs_epi16(x2, r0);
2089
3.51M
                x3 = _mm_maddubs_epi16(x3, r0);
2090
3.51M
                x4 = _mm_maddubs_epi16(x4, r0);
2091
3.51M
                x5 = _mm_maddubs_epi16(x5, r0);
2092
3.51M
                x2 = _mm_hadd_epi16(x2, x3);
2093
3.51M
                x4 = _mm_hadd_epi16(x4, x5);
2094
3.51M
                x2 = _mm_hadd_epi16(x2, x4);
2095
                /* give results back            */
2096
3.51M
                _mm_store_si128((__m128i *) &dst[x],x2);
2097
2098
3.51M
            }
2099
2.18M
            src += srcstride;
2100
2.18M
            dst += dststride;
2101
2.18M
        }
2102
217k
    }else if(!(width &3)){
2103
2104
351k
        for (y = 0; y < height; y ++) {
2105
677k
            for(x=0;x<width;x+=4){
2106
            /* load data in register     */
2107
362k
            x1 = _mm_loadu_si128((__m128i *) &src[x-3]);
2108
362k
            x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
2109
362k
            x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
2110
362k
                    _mm_srli_si128(x1, 3));
2111
2112
            /*  PMADDUBSW then PMADDW     */
2113
362k
            x2 = _mm_maddubs_epi16(x2, r0);
2114
362k
            x3 = _mm_maddubs_epi16(x3, r0);
2115
362k
            x2 = _mm_hadd_epi16(x2, x3);
2116
362k
            x2 = _mm_hadd_epi16(x2, x2);
2117
2118
            /* give results back            */
2119
362k
            _mm_storel_epi64((__m128i *) &dst[x], x2);
2120
362k
            }
2121
2122
314k
            src += srcstride;
2123
314k
            dst += dststride;
2124
314k
        }
2125
36.4k
    }else{
2126
0
        x5= _mm_setzero_si128();
2127
#if MASKMOVE
2128
        x3= _mm_set_epi32(0,0,0,-1);
2129
#endif
2130
0
        for (y = 0; y < height; y ++) {
2131
0
            for(x=0;x<width;x+=4){
2132
            /* load data in register     */
2133
0
            x1 = _mm_loadu_si128((__m128i *) &src[x-3]);
2134
0
            x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
2135
2136
2137
2138
            /*  PMADDUBSW then PMADDW     */
2139
0
            x2 = _mm_maddubs_epi16(x2, r0);
2140
0
            x2 = _mm_hadd_epi16(x2,x5 );
2141
0
            x2 = _mm_hadd_epi16(x2,x5 );
2142
2143
            /* give results back            */
2144
            //_mm_storel_epi64((__m128i *) &dst[x], x2);
2145
#if MASKMOVE
2146
            _mm_maskmoveu_si128(x2,x3,(char *) (dst+x));
2147
#else
2148
0
            *((uint16_t*)(dst+x)) = _mm_cvtsi128_si32(x2);
2149
0
#endif
2150
0
            }
2151
2152
0
            src += srcstride;
2153
0
            dst += dststride;
2154
0
        }
2155
0
    }
2156
2157
253k
}
2158
#ifndef __native_client__
2159
/*
2160
 * @TODO : Valgrind to see if it's useful to use SSE or wait for AVX2 implementation
2161
 */
2162
void ff_hevc_put_hevc_qpel_h_1_10_sse(int16_t *dst, ptrdiff_t dststride,
2163
                                      const uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
2164
0
        int16_t* mcbuffer) {
2165
0
    int x, y;
2166
0
    uint16_t *src = (uint16_t*)_src;
2167
0
    ptrdiff_t srcstride = _srcstride>>1;
2168
0
    __m128i x0, x1, x2, x3, r0;
2169
2170
0
    r0 = _mm_set_epi16(0, 1, -5, 17, 58, -10, 4, -1);
2171
0
    x0= _mm_setzero_si128();
2172
0
    x3= _mm_set_epi32(0,0,0,-1);
2173
0
    for (y = 0; y < height; y ++) {
2174
0
        for(x=0;x<width;x+=2){
2175
0
            x1 = _mm_loadu_si128((__m128i *) &src[x-3]);
2176
0
            x2 = _mm_srli_si128(x1,2); //last 16bit not used so 1 load can be used for 2 dst
2177
2178
0
            x1 = _mm_madd_epi16(x1,r0);
2179
0
            x2 = _mm_madd_epi16(x2,r0);
2180
2181
0
            x1 = _mm_hadd_epi32(x1,x2);
2182
0
            x1 = _mm_hadd_epi32(x1,x0);
2183
0
            x1= _mm_srai_epi32(x1,2); //>>BIT_DEPTH-8
2184
0
            x1= _mm_packs_epi32(x1,x0);
2185
         //   dst[x]= _mm_extract_epi16(x1,0);
2186
0
            _mm_maskmoveu_si128(x1,x3,(char *) (dst+x));
2187
0
        }
2188
0
        src += srcstride;
2189
0
        dst += dststride;
2190
0
    }
2191
2192
0
}
2193
#endif
2194
2195
2196
void ff_hevc_put_hevc_qpel_h_2_8_sse(int16_t *dst, ptrdiff_t dststride,
2197
                                     const uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
2198
155k
        int16_t* mcbuffer) {
2199
155k
    int x, y;
2200
155k
    const uint8_t *src = _src;
2201
155k
    ptrdiff_t srcstride = _srcstride / sizeof(uint8_t);
2202
155k
    __m128i x1, r0, x2, x3, x4, x5;
2203
2204
155k
    r0 = _mm_set_epi8(-1, 4, -11, 40, 40, -11, 4, -1, -1, 4, -11, 40, 40, -11,
2205
155k
            4, -1);
2206
2207
    /* LOAD src from memory to registers to limit memory bandwidth */
2208
155k
    if(!(width - 15)){
2209
0
        for (y = 0; y < height; y++) {
2210
0
                    for (x = 0; x < width; x += 8) {
2211
                        /* load data in register     */
2212
0
                        x1 = _mm_loadu_si128((__m128i *) &src[x - 3]);
2213
0
                        x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
2214
0
                        x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
2215
0
                                _mm_srli_si128(x1, 3));
2216
0
                        x4 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 4),
2217
0
                                _mm_srli_si128(x1, 5));
2218
0
                        x5 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 6),
2219
0
                                _mm_srli_si128(x1, 7));
2220
2221
                        /*  PMADDUBSW then PMADDW     */
2222
0
                        x2 = _mm_maddubs_epi16(x2, r0);
2223
0
                        x3 = _mm_maddubs_epi16(x3, r0);
2224
0
                        x4 = _mm_maddubs_epi16(x4, r0);
2225
0
                        x5 = _mm_maddubs_epi16(x5, r0);
2226
0
                        x2 = _mm_hadd_epi16(x2, x3);
2227
0
                        x4 = _mm_hadd_epi16(x4, x5);
2228
0
                        x2 = _mm_hadd_epi16(x2, x4);
2229
                        /* give results back            */
2230
0
                        _mm_store_si128((__m128i *) &dst[x],x2);
2231
0
                    }
2232
0
                    src += srcstride;
2233
0
                    dst += dststride;
2234
0
                }
2235
2236
155k
    }else{
2237
2238
1.61M
        for (y = 0; y < height; y ++) {
2239
5.75M
            for(x=0;x<width;x+=4){
2240
            /* load data in register     */
2241
4.30M
            x1 = _mm_loadu_si128((__m128i *) &src[x-3]);
2242
2243
4.30M
            x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
2244
4.30M
            x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
2245
4.30M
                    _mm_srli_si128(x1, 3));
2246
2247
2248
            /*  PMADDUBSW then PMADDW     */
2249
4.30M
            x2 = _mm_maddubs_epi16(x2, r0);
2250
4.30M
            x3 = _mm_maddubs_epi16(x3, r0);
2251
4.30M
            x2 = _mm_hadd_epi16(x2, x3);
2252
4.30M
            x2 = _mm_hadd_epi16(x2, _mm_setzero_si128());
2253
2254
            /* give results back            */
2255
4.30M
            _mm_storel_epi64((__m128i *) &dst[x], x2);
2256
2257
4.30M
            }
2258
1.45M
            src += srcstride;
2259
1.45M
            dst += dststride;
2260
1.45M
        }
2261
155k
    }
2262
2263
155k
}
2264
2265
#if 0
2266
static void ff_hevc_put_hevc_qpel_h_2_sse(int16_t *dst, ptrdiff_t dststride,
2267
                                          const uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
2268
        int16_t* mcbuffer) {
2269
    int x, y;
2270
    uint8_t *src = _src;
2271
    ptrdiff_t srcstride = _srcstride / sizeof(uint8_t);
2272
    __m128i x1, r0, x2, x3, x4, x5;
2273
2274
    r0 = _mm_set_epi8(-1, 4, -11, 40, 40, -11, 4, -1, -1, 4, -11, 40, 40, -11,
2275
            4, -1);
2276
2277
    /* LOAD src from memory to registers to limit memory bandwidth */
2278
    if(!(width & 7)){
2279
        for (y = 0; y < height; y++) {
2280
                    for (x = 0; x < width; x += 8) {
2281
                        /* load data in register     */
2282
                        x1 = _mm_loadu_si128((__m128i *) &src[x - 3]);
2283
                        x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
2284
                        x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
2285
                                _mm_srli_si128(x1, 3));
2286
                        x4 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 4),
2287
                                _mm_srli_si128(x1, 5));
2288
                        x5 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 6),
2289
                                _mm_srli_si128(x1, 7));
2290
2291
                        /*  PMADDUBSW then PMADDW     */
2292
                        x2 = _mm_maddubs_epi16(x2, r0);
2293
                        x3 = _mm_maddubs_epi16(x3, r0);
2294
                        x4 = _mm_maddubs_epi16(x4, r0);
2295
                        x5 = _mm_maddubs_epi16(x5, r0);
2296
                        x2 = _mm_hadd_epi16(x2, x3);
2297
                        x4 = _mm_hadd_epi16(x4, x5);
2298
                        x2 = _mm_hadd_epi16(x2, x4);
2299
                        /* give results back            */
2300
                        _mm_store_si128((__m128i *) &dst[x],x2);
2301
                    }
2302
                    src += srcstride;
2303
                    dst += dststride;
2304
                }
2305
2306
    }else{
2307
2308
        for (y = 0; y < height; y ++) {
2309
            for(x=0;x<width;x+=4){
2310
            /* load data in register     */
2311
            x1 = _mm_loadu_si128((__m128i *) &src[x-3]);
2312
2313
            x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
2314
            x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
2315
                    _mm_srli_si128(x1, 3));
2316
2317
2318
            /*  PMADDUBSW then PMADDW     */
2319
            x2 = _mm_maddubs_epi16(x2, r0);
2320
            x3 = _mm_maddubs_epi16(x3, r0);
2321
            x2 = _mm_hadd_epi16(x2, x3);
2322
            x2 = _mm_hadd_epi16(x2, _mm_setzero_si128());
2323
2324
            /* give results back            */
2325
            _mm_storel_epi64((__m128i *) &dst[x], x2);
2326
2327
            }
2328
            src += srcstride;
2329
            dst += dststride;
2330
        }
2331
    }
2332
2333
}
2334
static void ff_hevc_put_hevc_qpel_h_3_sse(int16_t *dst, ptrdiff_t dststride,
2335
                                          const uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
2336
        int16_t* mcbuffer) {
2337
    int x, y;
2338
    uint8_t *src = _src;
2339
    ptrdiff_t srcstride = _srcstride / sizeof(uint8_t);
2340
    __m128i x1, r0, x2, x3, x4, x5;
2341
2342
    r0 = _mm_set_epi8(-1, 4, -10, 58, 17, -5, 1, 0, -1, 4, -10, 58, 17, -5, 1,
2343
            0);
2344
2345
    if(!(width & 7)){
2346
        for (y = 0; y < height; y++) {
2347
            for (x = 0; x < width; x += 8) {
2348
                /* load data in register     */
2349
                x1 = _mm_loadu_si128((__m128i *) &src[x - 2]);
2350
                x1 = _mm_slli_si128(x1, 1);
2351
                x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
2352
                x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
2353
                        _mm_srli_si128(x1, 3));
2354
                x4 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 4),
2355
                        _mm_srli_si128(x1, 5));
2356
                x5 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 6),
2357
                        _mm_srli_si128(x1, 7));
2358
2359
                /*  PMADDUBSW then PMADDW     */
2360
                x2 = _mm_maddubs_epi16(x2, r0);
2361
                x3 = _mm_maddubs_epi16(x3, r0);
2362
                x4 = _mm_maddubs_epi16(x4, r0);
2363
                x5 = _mm_maddubs_epi16(x5, r0);
2364
                x2 = _mm_hadd_epi16(x2, x3);
2365
                x4 = _mm_hadd_epi16(x4, x5);
2366
                x2 = _mm_hadd_epi16(x2, x4);
2367
                /* give results back            */
2368
                _mm_store_si128((__m128i *) &dst[x],
2369
                        _mm_srli_si128(x2, BIT_DEPTH - 8));
2370
            }
2371
            src += srcstride;
2372
            dst += dststride;
2373
        }
2374
    }else{
2375
        for (y = 0; y < height; y ++) {
2376
            for(x=0;x<width;x+=4){
2377
                /* load data in register     */
2378
                x1 = _mm_loadu_si128((__m128i *) &src[x-2]);
2379
                x1 = _mm_slli_si128(x1, 1);
2380
                x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
2381
                x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
2382
                        _mm_srli_si128(x1, 3));
2383
2384
                /*  PMADDUBSW then PMADDW     */
2385
                x2 = _mm_maddubs_epi16(x2, r0);
2386
                x3 = _mm_maddubs_epi16(x3, r0);
2387
                x2 = _mm_hadd_epi16(x2, x3);
2388
                x2 = _mm_hadd_epi16(x2, _mm_setzero_si128());
2389
                x2 = _mm_srli_epi16(x2, BIT_DEPTH - 8);
2390
                /* give results back            */
2391
                _mm_storel_epi64((__m128i *) &dst[x], x2);
2392
2393
            }
2394
            src += srcstride;
2395
            dst += dststride;
2396
        }
2397
    }
2398
}
2399
#endif
2400
2401
void ff_hevc_put_hevc_qpel_h_3_8_sse(int16_t *dst, ptrdiff_t dststride,
2402
                                     const uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
2403
219k
        int16_t* mcbuffer) {
2404
219k
    int x, y;
2405
219k
    const uint8_t *src = _src;
2406
219k
    ptrdiff_t srcstride = _srcstride / sizeof(uint8_t);
2407
219k
    __m128i x1, r0, x2, x3, x4, x5;
2408
2409
219k
    r0 = _mm_set_epi8(-1, 4, -10, 58, 17, -5, 1, 0, -1, 4, -10, 58, 17, -5, 1,
2410
219k
            0);
2411
2412
219k
    if(!(width & 7)){
2413
2.04M
        for (y = 0; y < height; y++) {
2414
5.06M
            for (x = 0; x < width; x += 8) {
2415
                /* load data in register     */
2416
3.19M
                x1 = _mm_loadu_si128((__m128i *) &src[x - 2]);
2417
3.19M
                x1 = _mm_slli_si128(x1, 1);
2418
3.19M
                x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
2419
3.19M
                x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
2420
3.19M
                        _mm_srli_si128(x1, 3));
2421
3.19M
                x4 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 4),
2422
3.19M
                        _mm_srli_si128(x1, 5));
2423
3.19M
                x5 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 6),
2424
3.19M
                        _mm_srli_si128(x1, 7));
2425
2426
                /*  PMADDUBSW then PMADDW     */
2427
3.19M
                x2 = _mm_maddubs_epi16(x2, r0);
2428
3.19M
                x3 = _mm_maddubs_epi16(x3, r0);
2429
3.19M
                x4 = _mm_maddubs_epi16(x4, r0);
2430
3.19M
                x5 = _mm_maddubs_epi16(x5, r0);
2431
3.19M
                x2 = _mm_hadd_epi16(x2, x3);
2432
3.19M
                x4 = _mm_hadd_epi16(x4, x5);
2433
3.19M
                x2 = _mm_hadd_epi16(x2, x4);
2434
                /* give results back            */
2435
3.19M
                _mm_store_si128((__m128i *) &dst[x],x2);
2436
3.19M
            }
2437
1.86M
            src += srcstride;
2438
1.86M
            dst += dststride;
2439
1.86M
        }
2440
182k
    }else{
2441
355k
        for (y = 0; y < height; y ++) {
2442
695k
            for(x=0;x<width;x+=4){
2443
                /* load data in register     */
2444
376k
                x1 = _mm_loadu_si128((__m128i *) &src[x-2]);
2445
376k
                x1 = _mm_slli_si128(x1, 1);
2446
376k
                x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
2447
376k
                x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
2448
376k
                        _mm_srli_si128(x1, 3));
2449
2450
                /*  PMADDUBSW then PMADDW     */
2451
376k
                x2 = _mm_maddubs_epi16(x2, r0);
2452
376k
                x3 = _mm_maddubs_epi16(x3, r0);
2453
376k
                x2 = _mm_hadd_epi16(x2, x3);
2454
376k
                x2 = _mm_hadd_epi16(x2, _mm_setzero_si128());
2455
                /* give results back            */
2456
376k
                _mm_storel_epi64((__m128i *) &dst[x], x2);
2457
2458
376k
            }
2459
319k
            src += srcstride;
2460
319k
            dst += dststride;
2461
319k
        }
2462
36.7k
    }
2463
219k
}
2464
/**
2465
 for column MC treatment, we will calculate 8 pixels at the same time by multiplying the values
2466
 of each row.
2467
2468
 */
2469
void ff_hevc_put_hevc_qpel_v_1_8_sse(int16_t *dst, ptrdiff_t dststride,
2470
                                     const uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
2471
217k
        int16_t* mcbuffer) {
2472
217k
    int x, y;
2473
217k
    uint8_t *src = (uint8_t*) _src;
2474
217k
    ptrdiff_t srcstride = _srcstride / sizeof(uint8_t);
2475
217k
    __m128i x1, x2, x3, x4, x5, x6, x7, x8, r0, r1, r2;
2476
217k
    __m128i t1, t2, t3, t4, t5, t6, t7, t8;
2477
217k
    r1 = _mm_set_epi16(0, 1, -5, 17, 58, -10, 4, -1);
2478
2479
217k
    if(!(width & 15)){
2480
41.3k
        x8 = _mm_setzero_si128();
2481
761k
        for (y = 0; y < height; y++) {
2482
1.71M
            for (x = 0; x < width; x += 16) {
2483
                /* check if memory needs to be reloaded */
2484
995k
                x1 = _mm_loadu_si128((__m128i *) &src[x - 3 * srcstride]);
2485
995k
                x2 = _mm_loadu_si128((__m128i *) &src[x - 2 * srcstride]);
2486
995k
                x3 = _mm_loadu_si128((__m128i *) &src[x - srcstride]);
2487
995k
                x4 = _mm_loadu_si128((__m128i *) &src[x]);
2488
995k
                x5 = _mm_loadu_si128((__m128i *) &src[x + srcstride]);
2489
995k
                x6 = _mm_loadu_si128((__m128i *) &src[x + 2 * srcstride]);
2490
995k
                x7 = _mm_loadu_si128((__m128i *) &src[x + 3 * srcstride]);
2491
2492
995k
                t1 = _mm_unpacklo_epi8(x1,x8);
2493
995k
                t2 = _mm_unpacklo_epi8(x2, x8);
2494
995k
                t3 = _mm_unpacklo_epi8(x3, x8);
2495
995k
                t4 = _mm_unpacklo_epi8(x4, x8);
2496
995k
                t5 = _mm_unpacklo_epi8(x5, x8);
2497
995k
                t6 = _mm_unpacklo_epi8(x6, x8);
2498
995k
                t7 = _mm_unpacklo_epi8(x7, x8);
2499
2500
995k
                x1 = _mm_unpackhi_epi8(x1,x8);
2501
995k
                x2 = _mm_unpackhi_epi8(x2, x8);
2502
995k
                x3 = _mm_unpackhi_epi8(x3, x8);
2503
995k
                x4 = _mm_unpackhi_epi8(x4, x8);
2504
995k
                x5 = _mm_unpackhi_epi8(x5, x8);
2505
995k
                x6 = _mm_unpackhi_epi8(x6, x8);
2506
995k
                x7 = _mm_unpackhi_epi8(x7, x8);
2507
2508
                /* multiply by correct value : */
2509
995k
                r0 = _mm_mullo_epi16(t1,
2510
995k
                        _mm_set1_epi16(_mm_extract_epi16(r1, 0)));
2511
995k
                r2 = _mm_mullo_epi16(x1,
2512
995k
                        _mm_set1_epi16(_mm_extract_epi16(r1, 0)));
2513
995k
                r0 = _mm_adds_epi16(r0,
2514
995k
                        _mm_mullo_epi16(t2,
2515
995k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 1))));
2516
995k
                r2 = _mm_adds_epi16(r2,
2517
995k
                        _mm_mullo_epi16(x2,
2518
995k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 1))));
2519
995k
                r0 = _mm_adds_epi16(r0,
2520
995k
                        _mm_mullo_epi16(t3,
2521
995k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 2))));
2522
995k
                r2 = _mm_adds_epi16(r2,
2523
995k
                        _mm_mullo_epi16(x3,
2524
995k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 2))));
2525
2526
995k
                r0 = _mm_adds_epi16(r0,
2527
995k
                        _mm_mullo_epi16(t4,
2528
995k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 3))));
2529
995k
                r2 = _mm_adds_epi16(r2,
2530
995k
                        _mm_mullo_epi16(x4,
2531
995k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 3))));
2532
2533
995k
                r0 = _mm_adds_epi16(r0,
2534
995k
                        _mm_mullo_epi16(t5,
2535
995k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 4))));
2536
995k
                r2 = _mm_adds_epi16(r2,
2537
995k
                        _mm_mullo_epi16(x5,
2538
995k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 4))));
2539
2540
995k
                r0 = _mm_adds_epi16(r0,
2541
995k
                        _mm_mullo_epi16(t6,
2542
995k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 5))));
2543
995k
                r2 = _mm_adds_epi16(r2,
2544
995k
                        _mm_mullo_epi16(x6,
2545
995k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 5))));
2546
2547
995k
                r0 = _mm_adds_epi16(r0,
2548
995k
                        _mm_mullo_epi16(t7,
2549
995k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 6))));
2550
995k
                r2 = _mm_adds_epi16(r2,
2551
995k
                        _mm_mullo_epi16(x7,
2552
995k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 6))));
2553
2554
2555
                /* give results back            */
2556
995k
                _mm_store_si128((__m128i *) &dst[x],r0);
2557
995k
                _mm_store_si128((__m128i *) &dst[x + 8],r2);
2558
995k
            }
2559
720k
            src += srcstride;
2560
720k
            dst += dststride;
2561
720k
        }
2562
2563
176k
    }else{
2564
176k
        x = 0;
2565
176k
        x8 = _mm_setzero_si128();
2566
176k
        t8 = _mm_setzero_si128();
2567
1.63M
        for (y = 0; y < height; y ++) {
2568
4.18M
            for(x=0;x<width;x+=4){
2569
                /* load data in register  */
2570
2.72M
                x1 = _mm_loadl_epi64((__m128i *) &src[x-(3 * srcstride)]);
2571
2.72M
                x2 = _mm_loadl_epi64((__m128i *) &src[x-(2 * srcstride)]);
2572
2.72M
                x3 = _mm_loadl_epi64((__m128i *) &src[x-srcstride]);
2573
2.72M
                x4 = _mm_loadl_epi64((__m128i *) &src[x]);
2574
2.72M
                x5 = _mm_loadl_epi64((__m128i *) &src[x+srcstride]);
2575
2.72M
                x6 = _mm_loadl_epi64((__m128i *) &src[x+(2 * srcstride)]);
2576
2.72M
                x7 = _mm_loadl_epi64((__m128i *) &src[x+(3 * srcstride)]);
2577
2578
2579
2580
2.72M
                x1 = _mm_unpacklo_epi8(x1, t8);
2581
2.72M
                x2 = _mm_unpacklo_epi8(x2, t8);
2582
2.72M
                x3 = _mm_unpacklo_epi8(x3, t8);
2583
2.72M
                x4 = _mm_unpacklo_epi8(x4, t8);
2584
2.72M
                x5 = _mm_unpacklo_epi8(x5, t8);
2585
2.72M
                x6 = _mm_unpacklo_epi8(x6, t8);
2586
2.72M
                x7 = _mm_unpacklo_epi8(x7, t8);
2587
2588
2589
2.72M
                r0 = _mm_mullo_epi16(x1, _mm_set1_epi16(_mm_extract_epi16(r1, 0)));
2590
2591
2.72M
                r0 = _mm_adds_epi16(r0,
2592
2.72M
                        _mm_mullo_epi16(x2,
2593
2.72M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 1))));
2594
2595
2596
2.72M
                r0 = _mm_adds_epi16(r0,
2597
2.72M
                        _mm_mullo_epi16(x3,
2598
2.72M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 2))));
2599
2600
2.72M
                r0 = _mm_adds_epi16(r0,
2601
2.72M
                        _mm_mullo_epi16(x4,
2602
2.72M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 3))));
2603
2604
2.72M
                r0 = _mm_adds_epi16(r0,
2605
2.72M
                        _mm_mullo_epi16(x5,
2606
2.72M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 4))));
2607
2608
2609
2.72M
                r0 = _mm_adds_epi16(r0,
2610
2.72M
                        _mm_mullo_epi16(x6,
2611
2.72M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 5))));
2612
2613
2614
2.72M
                r0 = _mm_adds_epi16(r0,
2615
2.72M
                        _mm_mullo_epi16(x7,
2616
2.72M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 6))));
2617
2618
                /* give results back            */
2619
2.72M
                _mm_storel_epi64((__m128i *) &dst[x], r0);
2620
2.72M
            }
2621
1.45M
            src += srcstride;
2622
1.45M
            dst += dststride;
2623
1.45M
        }
2624
176k
    }
2625
217k
}
2626
2627
#if 0
2628
void ff_hevc_put_hevc_qpel_v_1_10_sse4(int16_t *dst, ptrdiff_t dststride,
2629
                                       const uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
2630
        int16_t* mcbuffer) {
2631
    int x, y;
2632
    uint16_t *src = (uint16_t*) _src;
2633
    ptrdiff_t srcstride = _srcstride >> 1;
2634
    __m128i x1, x2, x3, x4, x5, x6, x7, r1;
2635
    __m128i t1, t2, t3, t4, t5, t6, t7, t8;
2636
2637
        t7= _mm_set1_epi32(1);
2638
        t6= _mm_set1_epi32(-5);
2639
        t5= _mm_set1_epi32(17);
2640
        t4= _mm_set1_epi32(58);
2641
        t3= _mm_set1_epi32(-10);
2642
        t2= _mm_set1_epi32(4);
2643
        t1= _mm_set1_epi32(-1);
2644
        t8= _mm_setzero_si128();
2645
2646
        for (y = 0; y < height; y ++) {
2647
            for(x=0;x<width;x+=4){
2648
                /* load data in register  */
2649
                x1 = _mm_loadl_epi64((__m128i *) &src[x-(3 * srcstride)]);
2650
                x2 = _mm_loadl_epi64((__m128i *) &src[x-(2 * srcstride)]);
2651
                x3 = _mm_loadl_epi64((__m128i *) &src[x-srcstride]);
2652
                x4 = _mm_loadl_epi64((__m128i *) &src[x]);
2653
                x5 = _mm_loadl_epi64((__m128i *) &src[x+srcstride]);
2654
                x6 = _mm_loadl_epi64((__m128i *) &src[x+(2 * srcstride)]);
2655
                x7 = _mm_loadl_epi64((__m128i *) &src[x+(3 * srcstride)]);
2656
2657
2658
                x1 = _mm_unpacklo_epi16(x1, t8);
2659
                x2 = _mm_unpacklo_epi16(x2, t8);
2660
                x3 = _mm_unpacklo_epi16(x3, t8);
2661
                x4 = _mm_unpacklo_epi16(x4, t8);
2662
                x5 = _mm_unpacklo_epi16(x5, t8);
2663
                x6 = _mm_unpacklo_epi16(x6, t8);
2664
                x7 = _mm_unpacklo_epi16(x7, t8);
2665
2666
2667
                r1 = _mm_mullo_epi32(x1,t1);
2668
2669
                r1 = _mm_add_epi32(r1,
2670
                        _mm_mullo_epi32(x2,t2));
2671
2672
2673
                r1 = _mm_add_epi32(r1,
2674
                        _mm_mullo_epi32(x3,t3));
2675
2676
                r1 = _mm_add_epi32(r1,
2677
                        _mm_mullo_epi32(x4,t4));
2678
2679
                r1 = _mm_add_epi32(r1,
2680
                        _mm_mullo_epi32(x5,t5));
2681
2682
2683
                r1 = _mm_add_epi32(r1,
2684
                        _mm_mullo_epi32(x6,t6));
2685
2686
2687
                r1 = _mm_add_epi32(r1, _mm_mullo_epi32(x7,t7));
2688
                r1 = _mm_srai_epi32(r1,2); //bit depth - 8
2689
2690
2691
                r1 = _mm_packs_epi32(r1,t8);
2692
2693
                // give results back
2694
                _mm_storel_epi64((__m128i *) (dst + x), r1);
2695
            }
2696
            src += srcstride;
2697
            dst += dststride;
2698
        }
2699
2700
}
2701
#endif
2702
2703
2704
2705
void ff_hevc_put_hevc_qpel_v_2_8_sse(int16_t *dst, ptrdiff_t dststride,
2706
                                     const uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
2707
138k
        int16_t* mcbuffer) {
2708
138k
    int x, y;
2709
138k
    uint8_t *src = (uint8_t*) _src;
2710
138k
    ptrdiff_t srcstride = _srcstride / sizeof(uint8_t);
2711
138k
    __m128i x1, x2, x3, x4, x5, x6, x7, x8, r0, r1, r2;
2712
138k
    __m128i t1, t2, t3, t4, t5, t6, t7, t8;
2713
138k
    r1 = _mm_set_epi16(-1, 4, -11, 40, 40, -11, 4, -1);
2714
2715
138k
    if(!(width & 15)){
2716
427k
        for (y = 0; y < height; y++) {
2717
998k
            for (x = 0; x < width; x += 16) {
2718
593k
                r0 = _mm_setzero_si128();
2719
                /* check if memory needs to be reloaded */
2720
593k
                x1 = _mm_loadu_si128((__m128i *) &src[x - 3 * srcstride]);
2721
593k
                x2 = _mm_loadu_si128((__m128i *) &src[x - 2 * srcstride]);
2722
593k
                x3 = _mm_loadu_si128((__m128i *) &src[x - srcstride]);
2723
593k
                x4 = _mm_loadu_si128((__m128i *) &src[x]);
2724
593k
                x5 = _mm_loadu_si128((__m128i *) &src[x + srcstride]);
2725
593k
                x6 = _mm_loadu_si128((__m128i *) &src[x + 2 * srcstride]);
2726
593k
                x7 = _mm_loadu_si128((__m128i *) &src[x + 3 * srcstride]);
2727
593k
                x8 = _mm_loadu_si128((__m128i *) &src[x + 4 * srcstride]);
2728
2729
593k
                t1 = _mm_unpacklo_epi8(x1, r0);
2730
593k
                t2 = _mm_unpacklo_epi8(x2, r0);
2731
593k
                t3 = _mm_unpacklo_epi8(x3, r0);
2732
593k
                t4 = _mm_unpacklo_epi8(x4, r0);
2733
593k
                t5 = _mm_unpacklo_epi8(x5, r0);
2734
593k
                t6 = _mm_unpacklo_epi8(x6, r0);
2735
593k
                t7 = _mm_unpacklo_epi8(x7, r0);
2736
593k
                t8 = _mm_unpacklo_epi8(x8, r0);
2737
2738
593k
                x1 = _mm_unpackhi_epi8(x1, r0);
2739
593k
                x2 = _mm_unpackhi_epi8(x2, r0);
2740
593k
                x3 = _mm_unpackhi_epi8(x3, r0);
2741
593k
                x4 = _mm_unpackhi_epi8(x4, r0);
2742
593k
                x5 = _mm_unpackhi_epi8(x5, r0);
2743
593k
                x6 = _mm_unpackhi_epi8(x6, r0);
2744
593k
                x7 = _mm_unpackhi_epi8(x7, r0);
2745
593k
                x8 = _mm_unpackhi_epi8(x8, r0);
2746
2747
                /* multiply by correct value : */
2748
593k
                r0 = _mm_mullo_epi16(t1,
2749
593k
                        _mm_set1_epi16(_mm_extract_epi16(r1, 0)));
2750
593k
                r2 = _mm_mullo_epi16(x1,
2751
593k
                        _mm_set1_epi16(_mm_extract_epi16(r1, 0)));
2752
593k
                r0 = _mm_adds_epi16(r0,
2753
593k
                        _mm_mullo_epi16(t2,
2754
593k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 1))));
2755
593k
                r2 = _mm_adds_epi16(r2,
2756
593k
                        _mm_mullo_epi16(x2,
2757
593k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 1))));
2758
593k
                r0 = _mm_adds_epi16(r0,
2759
593k
                        _mm_mullo_epi16(t3,
2760
593k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 2))));
2761
593k
                r2 = _mm_adds_epi16(r2,
2762
593k
                        _mm_mullo_epi16(x3,
2763
593k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 2))));
2764
2765
593k
                r0 = _mm_adds_epi16(r0,
2766
593k
                        _mm_mullo_epi16(t4,
2767
593k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 3))));
2768
593k
                r2 = _mm_adds_epi16(r2,
2769
593k
                        _mm_mullo_epi16(x4,
2770
593k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 3))));
2771
2772
593k
                r0 = _mm_adds_epi16(r0,
2773
593k
                        _mm_mullo_epi16(t5,
2774
593k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 4))));
2775
593k
                r2 = _mm_adds_epi16(r2,
2776
593k
                        _mm_mullo_epi16(x5,
2777
593k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 4))));
2778
2779
593k
                r0 = _mm_adds_epi16(r0,
2780
593k
                        _mm_mullo_epi16(t6,
2781
593k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 5))));
2782
593k
                r2 = _mm_adds_epi16(r2,
2783
593k
                        _mm_mullo_epi16(x6,
2784
593k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 5))));
2785
2786
593k
                r0 = _mm_adds_epi16(r0,
2787
593k
                        _mm_mullo_epi16(t7,
2788
593k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 6))));
2789
593k
                r2 = _mm_adds_epi16(r2,
2790
593k
                        _mm_mullo_epi16(x7,
2791
593k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 6))));
2792
2793
593k
                r0 = _mm_adds_epi16(r0,
2794
593k
                        _mm_mullo_epi16(t8,
2795
593k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 7))));
2796
593k
                r2 = _mm_adds_epi16(r2,
2797
593k
                        _mm_mullo_epi16(x8,
2798
593k
                                _mm_set1_epi16(_mm_extract_epi16(r1, 7))));
2799
2800
                /* give results back            */
2801
593k
                _mm_store_si128((__m128i *) &dst[x],r0);
2802
593k
                _mm_store_si128((__m128i *) &dst[x + 8],r2);
2803
593k
            }
2804
405k
            src += srcstride;
2805
405k
            dst += dststride;
2806
405k
        }
2807
115k
    }else{
2808
115k
        x = 0;
2809
1.01M
        for (y = 0; y < height; y ++) {
2810
2.56M
            for(x=0;x<width;x+=4){
2811
1.66M
                r0 = _mm_setzero_si128();
2812
                /* load data in register  */
2813
1.66M
                x1 = _mm_loadl_epi64((__m128i *) &src[x - 3 * srcstride]);
2814
1.66M
                x2 = _mm_loadl_epi64((__m128i *) &src[x-2 * srcstride]);
2815
1.66M
                x3 = _mm_loadl_epi64((__m128i *) &src[x-srcstride]);
2816
1.66M
                x4 = _mm_loadl_epi64((__m128i *) &src[x]);
2817
1.66M
                x5 = _mm_loadl_epi64((__m128i *) &src[x+srcstride]);
2818
1.66M
                x6 = _mm_loadl_epi64((__m128i *) &src[x+2 * srcstride]);
2819
1.66M
                x7 = _mm_loadl_epi64((__m128i *) &src[x+3 * srcstride]);
2820
1.66M
                x8 = _mm_loadl_epi64((__m128i *) &src[x + 4 * srcstride]);
2821
2822
1.66M
                x1 = _mm_unpacklo_epi8(x1,r0);
2823
1.66M
                x2 = _mm_unpacklo_epi8(x2, r0);
2824
1.66M
                x3 = _mm_unpacklo_epi8(x3, r0);
2825
1.66M
                x4 = _mm_unpacklo_epi8(x4, r0);
2826
1.66M
                x5 = _mm_unpacklo_epi8(x5, r0);
2827
1.66M
                x6 = _mm_unpacklo_epi8(x6, r0);
2828
1.66M
                x7 = _mm_unpacklo_epi8(x7, r0);
2829
1.66M
                x8 = _mm_unpacklo_epi8(x8, r0);
2830
2831
2832
1.66M
                r0 = _mm_mullo_epi16(x1, _mm_set1_epi16(_mm_extract_epi16(r1, 0)));
2833
2834
1.66M
                r0 = _mm_adds_epi16(r0,
2835
1.66M
                        _mm_mullo_epi16(x2,
2836
1.66M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 1))));
2837
2838
2839
1.66M
                r0 = _mm_adds_epi16(r0,
2840
1.66M
                        _mm_mullo_epi16(x3,
2841
1.66M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 2))));
2842
2843
2844
1.66M
                r0 = _mm_adds_epi16(r0,
2845
1.66M
                        _mm_mullo_epi16(x4,
2846
1.66M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 3))));
2847
2848
2849
1.66M
                r0 = _mm_adds_epi16(r0,
2850
1.66M
                        _mm_mullo_epi16(x5,
2851
1.66M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 4))));
2852
2853
2854
1.66M
                r0 = _mm_adds_epi16(r0,
2855
1.66M
                        _mm_mullo_epi16(x6,
2856
1.66M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 5))));
2857
2858
2859
1.66M
                r0 = _mm_adds_epi16(r0,
2860
1.66M
                        _mm_mullo_epi16(x7,
2861
1.66M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 6))));
2862
2863
2864
1.66M
                r0 = _mm_adds_epi16(r0,
2865
1.66M
                        _mm_mullo_epi16(x8,
2866
1.66M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 7))));
2867
2868
2869
                /* give results back            */
2870
1.66M
                _mm_storel_epi64((__m128i *) &dst[x], r0);
2871
2872
1.66M
            }
2873
894k
            src += srcstride;
2874
894k
            dst += dststride;
2875
894k
        }
2876
115k
    }
2877
138k
}
2878
2879
#if 0
2880
void ff_hevc_put_hevc_qpel_v_2_10_sse(int16_t *dst, ptrdiff_t dststride,
2881
                                      cosnt uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
2882
        int16_t* mcbuffer) {
2883
    int x, y;
2884
    uint16_t *src = (uint16_t*) _src;
2885
    ptrdiff_t srcstride = _srcstride >> 1;
2886
    __m128i x1, x2, x3, x4, x5, x6, x7, x8, r0, r1, r2;
2887
    __m128i t1, t2, t3, t4, t5, t6, t7, t8;
2888
    r1 = _mm_set_epi16(-1, 4, -11, 40, 40, -11, 4, -1);
2889
2890
    t1= _mm_set1_epi32(-1);
2891
    t2= _mm_set1_epi32(4);
2892
    t3= _mm_set1_epi32(-11);
2893
    t4= _mm_set1_epi32(40);
2894
    t5= _mm_set1_epi32(40);
2895
    t6= _mm_set1_epi32(-11);
2896
    t7= _mm_set1_epi32(4);
2897
    t8= _mm_set1_epi32(-1);
2898
2899
    {
2900
        x = 0;
2901
        r0 = _mm_setzero_si128();
2902
        for (y = 0; y < height; y ++) {
2903
            for(x=0;x<width;x+=4){
2904
2905
                /* load data in register  */
2906
                x1 = _mm_loadl_epi64((__m128i *) &src[x - 3 * srcstride]);
2907
                x2 = _mm_loadl_epi64((__m128i *) &src[x-2 * srcstride]);
2908
                x3 = _mm_loadl_epi64((__m128i *) &src[x-srcstride]);
2909
                x4 = _mm_loadl_epi64((__m128i *) &src[x]);
2910
                x5 = _mm_loadl_epi64((__m128i *) &src[x+srcstride]);
2911
                x6 = _mm_loadl_epi64((__m128i *) &src[x+2 * srcstride]);
2912
                x7 = _mm_loadl_epi64((__m128i *) &src[x+3 * srcstride]);
2913
                x8 = _mm_loadl_epi64((__m128i *) &src[x + 4 * srcstride]);
2914
2915
                x1 = _mm_unpacklo_epi16(x1, r0);
2916
                x2 = _mm_unpacklo_epi16(x2, r0);
2917
                x3 = _mm_unpacklo_epi16(x3, r0);
2918
                x4 = _mm_unpacklo_epi16(x4, r0);
2919
                x5 = _mm_unpacklo_epi16(x5, r0);
2920
                x6 = _mm_unpacklo_epi16(x6, r0);
2921
                x7 = _mm_unpacklo_epi16(x7, r0);
2922
                x8 = _mm_unpacklo_epi16(x8, r0);
2923
2924
2925
                r1 = _mm_mullo_epi32(x1, t1);
2926
2927
                r1 = _mm_add_epi32(r1,
2928
                        _mm_mullo_epi32(x2,t2));
2929
2930
2931
                r1 = _mm_add_epi32(r1,
2932
                        _mm_mullo_epi32(x3,t3));
2933
2934
2935
                r1 = _mm_add_epi32(r1,
2936
                        _mm_mullo_epi32(x4,t4));
2937
2938
2939
                r1 = _mm_add_epi32(r1,
2940
                        _mm_mullo_epi32(x5,t5));
2941
2942
2943
                r1 = _mm_add_epi32(r1,
2944
                        _mm_mullo_epi32(x6,t6));
2945
2946
2947
                r1 = _mm_add_epi32(r1,
2948
                        _mm_mullo_epi32(x7,t7));
2949
2950
2951
                r1 = _mm_add_epi32(r1,
2952
                        _mm_mullo_epi32(x8,t8));
2953
2954
2955
                r1= _mm_srai_epi32(r1,2); //bit depth - 8
2956
2957
                r1= _mm_packs_epi32(r1,t8);
2958
2959
                /* give results back            */
2960
                _mm_storel_epi64((__m128i *) (dst+x), r1);
2961
2962
            }
2963
            src += srcstride;
2964
            dst += dststride;
2965
        }
2966
    }
2967
}
2968
#endif
2969
2970
#if 0
2971
static  void ff_hevc_put_hevc_qpel_v_3_sse(int16_t *dst, ptrdiff_t dststride,
2972
                                           const uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
2973
        int16_t* mcbuffer) {
2974
    int x, y;
2975
    uint8_t *src = (uint8_t*) _src;
2976
    ptrdiff_t srcstride = _srcstride / sizeof(uint8_t);
2977
    __m128i x1, x2, x3, x4, x5, x6, x7, x8, r0, r1, r2;
2978
    __m128i t2, t3, t4, t5, t6, t7, t8;
2979
    r1 = _mm_set_epi16(-1, 4, -10, 58, 17, -5, 1, 0);
2980
2981
    if(!(width & 15)){
2982
        for (y = 0; y < height; y++) {
2983
                    for (x = 0; x < width; x += 16) {
2984
                        /* check if memory needs to be reloaded */
2985
                        x1 = _mm_setzero_si128();
2986
                        x2 = _mm_loadu_si128((__m128i *) &src[x - 2 * srcstride]);
2987
                        x3 = _mm_loadu_si128((__m128i *) &src[x - srcstride]);
2988
                        x4 = _mm_loadu_si128((__m128i *) &src[x]);
2989
                        x5 = _mm_loadu_si128((__m128i *) &src[x + srcstride]);
2990
                        x6 = _mm_loadu_si128((__m128i *) &src[x + 2 * srcstride]);
2991
                        x7 = _mm_loadu_si128((__m128i *) &src[x + 3 * srcstride]);
2992
                        x8 = _mm_loadu_si128((__m128i *) &src[x + 4 * srcstride]);
2993
2994
                        t2 = _mm_unpacklo_epi8(x2, x1);
2995
                        t3 = _mm_unpacklo_epi8(x3, x1);
2996
                        t4 = _mm_unpacklo_epi8(x4, x1);
2997
                        t5 = _mm_unpacklo_epi8(x5, x1);
2998
                        t6 = _mm_unpacklo_epi8(x6, x1);
2999
                        t7 = _mm_unpacklo_epi8(x7, x1);
3000
                        t8 = _mm_unpacklo_epi8(x8, x1);
3001
3002
                        x2 = _mm_unpackhi_epi8(x2, x1);
3003
                        x3 = _mm_unpackhi_epi8(x3, x1);
3004
                        x4 = _mm_unpackhi_epi8(x4, x1);
3005
                        x5 = _mm_unpackhi_epi8(x5, x1);
3006
                        x6 = _mm_unpackhi_epi8(x6, x1);
3007
                        x7 = _mm_unpackhi_epi8(x7, x1);
3008
                        x8 = _mm_unpackhi_epi8(x8, x1);
3009
3010
                        /* multiply by correct value : */
3011
                        r0 = _mm_mullo_epi16(t2,
3012
                                _mm_set1_epi16(_mm_extract_epi16(r1, 1)));
3013
                        r2 = _mm_mullo_epi16(x2,
3014
                                _mm_set1_epi16(_mm_extract_epi16(r1, 1)));
3015
3016
                        r0 = _mm_adds_epi16(r0,
3017
                                _mm_mullo_epi16(t3,
3018
                                        _mm_set1_epi16(_mm_extract_epi16(r1, 2))));
3019
                        r2 = _mm_adds_epi16(r2,
3020
                                _mm_mullo_epi16(x3,
3021
                                        _mm_set1_epi16(_mm_extract_epi16(r1, 2))));
3022
3023
                        r0 = _mm_adds_epi16(r0,
3024
                                _mm_mullo_epi16(t4,
3025
                                        _mm_set1_epi16(_mm_extract_epi16(r1, 3))));
3026
                        r2 = _mm_adds_epi16(r2,
3027
                                _mm_mullo_epi16(x4,
3028
                                        _mm_set1_epi16(_mm_extract_epi16(r1, 3))));
3029
3030
                        r0 = _mm_adds_epi16(r0,
3031
                                _mm_mullo_epi16(t5,
3032
                                        _mm_set1_epi16(_mm_extract_epi16(r1, 4))));
3033
                        r2 = _mm_adds_epi16(r2,
3034
                                _mm_mullo_epi16(x5,
3035
                                        _mm_set1_epi16(_mm_extract_epi16(r1, 4))));
3036
3037
                        r0 = _mm_adds_epi16(r0,
3038
                                _mm_mullo_epi16(t6,
3039
                                        _mm_set1_epi16(_mm_extract_epi16(r1, 5))));
3040
                        r2 = _mm_adds_epi16(r2,
3041
                                _mm_mullo_epi16(x6,
3042
                                        _mm_set1_epi16(_mm_extract_epi16(r1, 5))));
3043
3044
                        r0 = _mm_adds_epi16(r0,
3045
                                _mm_mullo_epi16(t7,
3046
                                        _mm_set1_epi16(_mm_extract_epi16(r1, 6))));
3047
                        r2 = _mm_adds_epi16(r2,
3048
                                _mm_mullo_epi16(x7,
3049
                                        _mm_set1_epi16(_mm_extract_epi16(r1, 6))));
3050
3051
                        r0 = _mm_adds_epi16(r0,
3052
                                _mm_mullo_epi16(t8,
3053
                                        _mm_set1_epi16(_mm_extract_epi16(r1, 7))));
3054
                        r2 = _mm_adds_epi16(r2,
3055
                                _mm_mullo_epi16(x8,
3056
                                        _mm_set1_epi16(_mm_extract_epi16(r1, 7))));
3057
3058
                        /* give results back            */
3059
                        _mm_store_si128((__m128i *) &dst[x],
3060
                                _mm_srli_epi16(r0, BIT_DEPTH - 8));
3061
                        _mm_store_si128((__m128i *) &dst[x + 8],
3062
                                _mm_srli_epi16(r2, BIT_DEPTH - 8));
3063
                    }
3064
                    src += srcstride;
3065
                    dst += dststride;
3066
                }
3067
    }else{
3068
        x = 0;
3069
                for (y = 0; y < height; y ++) {
3070
                    for(x=0;x<width;x+=4){
3071
                    r0 = _mm_set1_epi16(0);
3072
                    /* load data in register  */
3073
                    //x1 = _mm_setzero_si128();
3074
                    x2 = _mm_loadl_epi64((__m128i *) &src[x-2 * srcstride]);
3075
                    x3 = _mm_loadl_epi64((__m128i *) &src[x-srcstride]);
3076
                    x4 = _mm_loadl_epi64((__m128i *) &src[x]);
3077
                    x5 = _mm_loadl_epi64((__m128i *) &src[x+srcstride]);
3078
                    x6 = _mm_loadl_epi64((__m128i *) &src[x+2 * srcstride]);
3079
                    x7 = _mm_loadl_epi64((__m128i *) &src[x+3 * srcstride]);
3080
                    x8 = _mm_loadl_epi64((__m128i *) &src[x + 4 * srcstride]);
3081
3082
                    x1 = _mm_unpacklo_epi8(x1,r0);
3083
                    x2 = _mm_unpacklo_epi8(x2, r0);
3084
                    x3 = _mm_unpacklo_epi8(x3, r0);
3085
                    x4 = _mm_unpacklo_epi8(x4, r0);
3086
                    x5 = _mm_unpacklo_epi8(x5, r0);
3087
                    x6 = _mm_unpacklo_epi8(x6, r0);
3088
                    x7 = _mm_unpacklo_epi8(x7, r0);
3089
                    x8 = _mm_unpacklo_epi8(x8, r0);
3090
3091
3092
                    r0 = _mm_mullo_epi16(x2, _mm_set1_epi16(_mm_extract_epi16(r1, 1)));
3093
3094
3095
                    r0 = _mm_adds_epi16(r0,
3096
                            _mm_mullo_epi16(x3,
3097
                                    _mm_set1_epi16(_mm_extract_epi16(r1, 2))));
3098
3099
3100
                    r0 = _mm_adds_epi16(r0,
3101
                            _mm_mullo_epi16(x4,
3102
                                    _mm_set1_epi16(_mm_extract_epi16(r1, 3))));
3103
3104
3105
                    r0 = _mm_adds_epi16(r0,
3106
                            _mm_mullo_epi16(x5,
3107
                                    _mm_set1_epi16(_mm_extract_epi16(r1, 4))));
3108
3109
3110
                    r0 = _mm_adds_epi16(r0,
3111
                            _mm_mullo_epi16(x6,
3112
                                    _mm_set1_epi16(_mm_extract_epi16(r1, 5))));
3113
3114
3115
                    r0 = _mm_adds_epi16(r0,
3116
                            _mm_mullo_epi16(x7,
3117
                                    _mm_set1_epi16(_mm_extract_epi16(r1, 6))));
3118
3119
3120
                    r0 = _mm_adds_epi16(r0,
3121
                            _mm_mullo_epi16(x8,
3122
                                    _mm_set1_epi16(_mm_extract_epi16(r1, 7))));
3123
3124
3125
                    r0 = _mm_srli_epi16(r0, BIT_DEPTH - 8);
3126
                    /* give results back            */
3127
                    _mm_storel_epi64((__m128i *) &dst[x], r0);
3128
3129
                    }
3130
                    src += srcstride;
3131
                    dst += dststride;
3132
                }
3133
    }
3134
3135
}
3136
#endif
3137
3138
void ff_hevc_put_hevc_qpel_v_3_8_sse(int16_t *dst, ptrdiff_t dststride,
3139
                                     const uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
3140
251k
        int16_t* mcbuffer) {
3141
251k
    int x, y;
3142
251k
    uint8_t *src = (uint8_t*) _src;
3143
251k
    ptrdiff_t srcstride = _srcstride / sizeof(uint8_t);
3144
251k
    __m128i x1, x2, x3, x4, x5, x6, x7, x8, r0, r1, r2;
3145
251k
    __m128i t2, t3, t4, t5, t6, t7, t8;
3146
251k
    r1 = _mm_set_epi16(-1, 4, -10, 58, 17, -5, 1, 0);
3147
3148
251k
    if(!(width & 15)){
3149
703k
        for (y = 0; y < height; y++) {
3150
1.69M
            for (x = 0; x < width; x += 16) {
3151
                /* check if memory needs to be reloaded */
3152
1.02M
                x1 = _mm_setzero_si128();
3153
1.02M
                x2 = _mm_loadu_si128((__m128i *) &src[x - 2 * srcstride]);
3154
1.02M
                x3 = _mm_loadu_si128((__m128i *) &src[x - srcstride]);
3155
1.02M
                x4 = _mm_loadu_si128((__m128i *) &src[x]);
3156
1.02M
                x5 = _mm_loadu_si128((__m128i *) &src[x + srcstride]);
3157
1.02M
                x6 = _mm_loadu_si128((__m128i *) &src[x + 2 * srcstride]);
3158
1.02M
                x7 = _mm_loadu_si128((__m128i *) &src[x + 3 * srcstride]);
3159
1.02M
                x8 = _mm_loadu_si128((__m128i *) &src[x + 4 * srcstride]);
3160
3161
1.02M
                t2 = _mm_unpacklo_epi8(x2, x1);
3162
1.02M
                t3 = _mm_unpacklo_epi8(x3, x1);
3163
1.02M
                t4 = _mm_unpacklo_epi8(x4, x1);
3164
1.02M
                t5 = _mm_unpacklo_epi8(x5, x1);
3165
1.02M
                t6 = _mm_unpacklo_epi8(x6, x1);
3166
1.02M
                t7 = _mm_unpacklo_epi8(x7, x1);
3167
1.02M
                t8 = _mm_unpacklo_epi8(x8, x1);
3168
3169
1.02M
                x2 = _mm_unpackhi_epi8(x2, x1);
3170
1.02M
                x3 = _mm_unpackhi_epi8(x3, x1);
3171
1.02M
                x4 = _mm_unpackhi_epi8(x4, x1);
3172
1.02M
                x5 = _mm_unpackhi_epi8(x5, x1);
3173
1.02M
                x6 = _mm_unpackhi_epi8(x6, x1);
3174
1.02M
                x7 = _mm_unpackhi_epi8(x7, x1);
3175
1.02M
                x8 = _mm_unpackhi_epi8(x8, x1);
3176
3177
                /* multiply by correct value : */
3178
1.02M
                r0 = _mm_mullo_epi16(t2,
3179
1.02M
                        _mm_set1_epi16(_mm_extract_epi16(r1, 1)));
3180
1.02M
                r2 = _mm_mullo_epi16(x2,
3181
1.02M
                        _mm_set1_epi16(_mm_extract_epi16(r1, 1)));
3182
3183
1.02M
                r0 = _mm_adds_epi16(r0,
3184
1.02M
                        _mm_mullo_epi16(t3,
3185
1.02M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 2))));
3186
1.02M
                r2 = _mm_adds_epi16(r2,
3187
1.02M
                        _mm_mullo_epi16(x3,
3188
1.02M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 2))));
3189
3190
1.02M
                r0 = _mm_adds_epi16(r0,
3191
1.02M
                        _mm_mullo_epi16(t4,
3192
1.02M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 3))));
3193
1.02M
                r2 = _mm_adds_epi16(r2,
3194
1.02M
                        _mm_mullo_epi16(x4,
3195
1.02M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 3))));
3196
3197
1.02M
                r0 = _mm_adds_epi16(r0,
3198
1.02M
                        _mm_mullo_epi16(t5,
3199
1.02M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 4))));
3200
1.02M
                r2 = _mm_adds_epi16(r2,
3201
1.02M
                        _mm_mullo_epi16(x5,
3202
1.02M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 4))));
3203
3204
1.02M
                r0 = _mm_adds_epi16(r0,
3205
1.02M
                        _mm_mullo_epi16(t6,
3206
1.02M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 5))));
3207
1.02M
                r2 = _mm_adds_epi16(r2,
3208
1.02M
                        _mm_mullo_epi16(x6,
3209
1.02M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 5))));
3210
3211
1.02M
                r0 = _mm_adds_epi16(r0,
3212
1.02M
                        _mm_mullo_epi16(t7,
3213
1.02M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 6))));
3214
1.02M
                r2 = _mm_adds_epi16(r2,
3215
1.02M
                        _mm_mullo_epi16(x7,
3216
1.02M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 6))));
3217
3218
1.02M
                r0 = _mm_adds_epi16(r0,
3219
1.02M
                        _mm_mullo_epi16(t8,
3220
1.02M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 7))));
3221
1.02M
                r2 = _mm_adds_epi16(r2,
3222
1.02M
                        _mm_mullo_epi16(x8,
3223
1.02M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 7))));
3224
3225
                /* give results back            */
3226
1.02M
                _mm_store_si128((__m128i *) &dst[x],r0);
3227
1.02M
                _mm_store_si128((__m128i *) &dst[x + 8],r2);
3228
1.02M
            }
3229
666k
            src += srcstride;
3230
666k
            dst += dststride;
3231
666k
        }
3232
213k
    }else{
3233
213k
        x = 0;
3234
1.92M
        for (y = 0; y < height; y ++) {
3235
4.87M
            for(x=0;x<width;x+=4){
3236
3.16M
                r0 = _mm_set1_epi16(0);
3237
                /* load data in register  */
3238
3.16M
                x2 = _mm_loadl_epi64((__m128i *) &src[x-2 * srcstride]);
3239
3.16M
                x3 = _mm_loadl_epi64((__m128i *) &src[x-srcstride]);
3240
3.16M
                x4 = _mm_loadl_epi64((__m128i *) &src[x]);
3241
3.16M
                x5 = _mm_loadl_epi64((__m128i *) &src[x+srcstride]);
3242
3.16M
                x6 = _mm_loadl_epi64((__m128i *) &src[x+2 * srcstride]);
3243
3.16M
                x7 = _mm_loadl_epi64((__m128i *) &src[x+3 * srcstride]);
3244
3.16M
                x8 = _mm_loadl_epi64((__m128i *) &src[x + 4 * srcstride]);
3245
3246
3.16M
                x2 = _mm_unpacklo_epi8(x2, r0);
3247
3.16M
                x3 = _mm_unpacklo_epi8(x3, r0);
3248
3.16M
                x4 = _mm_unpacklo_epi8(x4, r0);
3249
3.16M
                x5 = _mm_unpacklo_epi8(x5, r0);
3250
3.16M
                x6 = _mm_unpacklo_epi8(x6, r0);
3251
3.16M
                x7 = _mm_unpacklo_epi8(x7, r0);
3252
3.16M
                x8 = _mm_unpacklo_epi8(x8, r0);
3253
3254
3.16M
                r0 = _mm_mullo_epi16(x2, _mm_set1_epi16(_mm_extract_epi16(r1, 1)));
3255
3256
3.16M
                r0 = _mm_adds_epi16(r0,
3257
3.16M
                        _mm_mullo_epi16(x3,
3258
3.16M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 2))));
3259
3260
3.16M
                r0 = _mm_adds_epi16(r0,
3261
3.16M
                        _mm_mullo_epi16(x4,
3262
3.16M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 3))));
3263
3264
3.16M
                r0 = _mm_adds_epi16(r0,
3265
3.16M
                        _mm_mullo_epi16(x5,
3266
3.16M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 4))));
3267
3268
3.16M
                r0 = _mm_adds_epi16(r0,
3269
3.16M
                        _mm_mullo_epi16(x6,
3270
3.16M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 5))));
3271
3272
3.16M
                r0 = _mm_adds_epi16(r0,
3273
3.16M
                        _mm_mullo_epi16(x7,
3274
3.16M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 6))));
3275
3276
3.16M
                r0 = _mm_adds_epi16(r0,
3277
3.16M
                        _mm_mullo_epi16(x8,
3278
3.16M
                                _mm_set1_epi16(_mm_extract_epi16(r1, 7))));
3279
3280
                /* give results back            */
3281
3.16M
                _mm_storel_epi64((__m128i *) &dst[x], r0);
3282
3283
3.16M
            }
3284
1.71M
            src += srcstride;
3285
1.71M
            dst += dststride;
3286
1.71M
        }
3287
213k
    }
3288
3289
251k
}
3290
3291
3292
#if 0
3293
void ff_hevc_put_hevc_qpel_v_3_10_sse(int16_t *dst, ptrdiff_t dststride,
3294
                                      const uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
3295
        int16_t* mcbuffer) {
3296
    int x, y;
3297
    uint16_t *src = (uint16_t*) _src;
3298
    ptrdiff_t srcstride = _srcstride >> 1;
3299
    __m128i x1, x2, x3, x4, x5, x6, x7, r0;
3300
    __m128i t1, t2, t3, t4, t5, t6, t7, t8;
3301
3302
    t7 = _mm_set1_epi32(-1);
3303
    t6 = _mm_set1_epi32(4);
3304
    t5 = _mm_set1_epi32(-10);
3305
    t4 = _mm_set1_epi32(58);
3306
    t3 = _mm_set1_epi32(17);
3307
    t2 = _mm_set1_epi32(-5);
3308
    t1 = _mm_set1_epi32(1);
3309
    t8= _mm_setzero_si128();
3310
    {
3311
3312
        for (y = 0; y < height; y ++) {
3313
            for(x=0;x<width;x+=4){
3314
                /* load data in register  */
3315
                x1 = _mm_loadl_epi64((__m128i *) &src[x-2 * srcstride]);
3316
                x2 = _mm_loadl_epi64((__m128i *) &src[x-srcstride]);
3317
                x3 = _mm_loadl_epi64((__m128i *) &src[x]);
3318
                x4 = _mm_loadl_epi64((__m128i *) &src[x+srcstride]);
3319
                x5 = _mm_loadl_epi64((__m128i *) &src[x+2 * srcstride]);
3320
                x6 = _mm_loadl_epi64((__m128i *) &src[x+3 * srcstride]);
3321
                x7 = _mm_loadl_epi64((__m128i *) &src[x + 4 * srcstride]);
3322
3323
                x1 = _mm_unpacklo_epi16(x1, t8);
3324
                x2 = _mm_unpacklo_epi16(x2, t8);
3325
                x3 = _mm_unpacklo_epi16(x3, t8);
3326
                x4 = _mm_unpacklo_epi16(x4, t8);
3327
                x5 = _mm_unpacklo_epi16(x5, t8);
3328
                x6 = _mm_unpacklo_epi16(x6, t8);
3329
                x7 = _mm_unpacklo_epi16(x7, t8);
3330
3331
                r0 = _mm_mullo_epi32(x1, t1);
3332
3333
                r0 = _mm_add_epi32(r0,
3334
                        _mm_mullo_epi32(x2,t2));
3335
3336
                r0 = _mm_add_epi32(r0,
3337
                        _mm_mullo_epi32(x3,t3));
3338
3339
                r0 = _mm_add_epi32(r0,
3340
                        _mm_mullo_epi32(x4,t4));
3341
3342
                r0 = _mm_add_epi32(r0,
3343
                        _mm_mullo_epi32(x5,t5));
3344
3345
                r0 = _mm_add_epi32(r0,
3346
                        _mm_mullo_epi32(x6,t6));
3347
3348
                r0 = _mm_add_epi32(r0,
3349
                        _mm_mullo_epi32(x7,t7));
3350
3351
                r0= _mm_srai_epi32(r0,2);
3352
3353
                r0= _mm_packs_epi32(r0,t8);
3354
3355
                /* give results back            */
3356
                _mm_storel_epi64((__m128i *) &dst[x], r0);
3357
3358
            }
3359
            src += srcstride;
3360
            dst += dststride;
3361
        }
3362
    }
3363
3364
}
3365
#endif
3366
3367
3368
3369
void ff_hevc_put_hevc_qpel_h_1_v_1_sse(int16_t *dst, ptrdiff_t dststride,
3370
                                       const uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
3371
257k
        int16_t* mcbuffer) {
3372
257k
    int x, y;
3373
257k
    uint8_t* src = (uint8_t*) _src;
3374
257k
    ptrdiff_t srcstride = _srcstride / sizeof(uint8_t);
3375
257k
    int16_t *tmp = mcbuffer;
3376
257k
    __m128i x1, x2, x3, x4, x5, x6, x7, rBuffer, rTemp, r0, r1;
3377
257k
    __m128i t1, t2, t3, t4, t5, t6, t7, t8;
3378
3379
257k
    src -= qpel_extra_before[1] * srcstride;
3380
257k
    r0 = _mm_set_epi8(0, 1, -5, 17, 58, -10, 4, -1, 0, 1, -5, 17, 58, -10, 4,
3381
257k
            -1);
3382
3383
    /* LOAD src from memory to registers to limit memory bandwidth */
3384
257k
    if (width == 4) {
3385
3386
515k
        for (y = 0; y < height + qpel_extra[1]; y += 2) {
3387
            /* load data in register     */
3388
451k
            x1 = _mm_loadu_si128((__m128i *) &src[-3]);
3389
451k
            src += srcstride;
3390
451k
            t1 = _mm_loadu_si128((__m128i *) &src[-3]);
3391
451k
            x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
3392
451k
            t2 = _mm_unpacklo_epi64(t1, _mm_srli_si128(t1, 1));
3393
451k
            x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
3394
451k
                    _mm_srli_si128(x1, 3));
3395
451k
            t3 = _mm_unpacklo_epi64(_mm_srli_si128(t1, 2),
3396
451k
                    _mm_srli_si128(t1, 3));
3397
3398
            /*  PMADDUBSW then PMADDW     */
3399
451k
            x2 = _mm_maddubs_epi16(x2, r0);
3400
451k
            t2 = _mm_maddubs_epi16(t2, r0);
3401
451k
            x3 = _mm_maddubs_epi16(x3, r0);
3402
451k
            t3 = _mm_maddubs_epi16(t3, r0);
3403
451k
            x2 = _mm_hadd_epi16(x2, x3);
3404
451k
            t2 = _mm_hadd_epi16(t2, t3);
3405
451k
            x2 = _mm_hadd_epi16(x2, _mm_set1_epi16(0));
3406
451k
            t2 = _mm_hadd_epi16(t2, _mm_set1_epi16(0));
3407
451k
            x2 = _mm_srli_epi16(x2, BIT_DEPTH - 8);
3408
451k
            t2 = _mm_srli_epi16(t2, BIT_DEPTH - 8);
3409
            /* give results back            */
3410
451k
            _mm_storel_epi64((__m128i *) &tmp[0], x2);
3411
3412
451k
            tmp += MAX_PB_SIZE;
3413
451k
            _mm_storel_epi64((__m128i *) &tmp[0], t2);
3414
3415
451k
            src += srcstride;
3416
451k
            tmp += MAX_PB_SIZE;
3417
451k
        }
3418
63.8k
    } else
3419
3.20M
        for (y = 0; y < height + qpel_extra[1]; y++) {
3420
7.76M
            for (x = 0; x < width; x += 8) {
3421
                /* load data in register     */
3422
4.75M
                x1 = _mm_loadu_si128((__m128i *) &src[x - 3]);
3423
4.75M
                x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
3424
4.75M
                x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
3425
4.75M
                        _mm_srli_si128(x1, 3));
3426
4.75M
                x4 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 4),
3427
4.75M
                        _mm_srli_si128(x1, 5));
3428
4.75M
                x5 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 6),
3429
4.75M
                        _mm_srli_si128(x1, 7));
3430
3431
                /*  PMADDUBSW then PMADDW     */
3432
4.75M
                x2 = _mm_maddubs_epi16(x2, r0);
3433
4.75M
                x3 = _mm_maddubs_epi16(x3, r0);
3434
4.75M
                x4 = _mm_maddubs_epi16(x4, r0);
3435
4.75M
                x5 = _mm_maddubs_epi16(x5, r0);
3436
4.75M
                x2 = _mm_hadd_epi16(x2, x3);
3437
4.75M
                x4 = _mm_hadd_epi16(x4, x5);
3438
4.75M
                x2 = _mm_hadd_epi16(x2, x4);
3439
4.75M
                x2 = _mm_srli_si128(x2, BIT_DEPTH - 8);
3440
3441
                /* give results back            */
3442
4.75M
                _mm_store_si128((__m128i *) &tmp[x], x2);
3443
3444
4.75M
            }
3445
3.01M
            src += srcstride;
3446
3.01M
            tmp += MAX_PB_SIZE;
3447
3.01M
        }
3448
3449
257k
    tmp = mcbuffer + qpel_extra_before[1] * MAX_PB_SIZE;
3450
257k
    srcstride = MAX_PB_SIZE;
3451
3452
    /* vertical treatment on temp table : tmp contains 16 bit values, so need to use 32 bit  integers
3453
     for register calculations */
3454
257k
    rTemp = _mm_set_epi16(0, 1, -5, 17, 58, -10, 4, -1);
3455
2.62M
    for (y = 0; y < height; y++) {
3456
6.13M
        for (x = 0; x < width; x += 8) {
3457
3458
3.76M
            x1 = _mm_load_si128((__m128i *) &tmp[x - 3 * srcstride]);
3459
3.76M
            x2 = _mm_load_si128((__m128i *) &tmp[x - 2 * srcstride]);
3460
3.76M
            x3 = _mm_load_si128((__m128i *) &tmp[x - srcstride]);
3461
3.76M
            x4 = _mm_load_si128((__m128i *) &tmp[x]);
3462
3.76M
            x5 = _mm_load_si128((__m128i *) &tmp[x + srcstride]);
3463
3.76M
            x6 = _mm_load_si128((__m128i *) &tmp[x + 2 * srcstride]);
3464
3.76M
            x7 = _mm_load_si128((__m128i *) &tmp[x + 3 * srcstride]);
3465
3466
3.76M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 0));
3467
3.76M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 1));
3468
3.76M
            t8 = _mm_mullo_epi16(x1, r0);
3469
3.76M
            rBuffer = _mm_mulhi_epi16(x1, r0);
3470
3.76M
            t7 = _mm_mullo_epi16(x2, r1);
3471
3.76M
            t1 = _mm_unpacklo_epi16(t8, rBuffer);
3472
3.76M
            x1 = _mm_unpackhi_epi16(t8, rBuffer);
3473
3474
3.76M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 2));
3475
3.76M
            rBuffer = _mm_mulhi_epi16(x2, r1);
3476
3.76M
            t8 = _mm_mullo_epi16(x3, r0);
3477
3.76M
            t2 = _mm_unpacklo_epi16(t7, rBuffer);
3478
3.76M
            x2 = _mm_unpackhi_epi16(t7, rBuffer);
3479
3480
3.76M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 3));
3481
3.76M
            rBuffer = _mm_mulhi_epi16(x3, r0);
3482
3.76M
            t7 = _mm_mullo_epi16(x4, r1);
3483
3.76M
            t3 = _mm_unpacklo_epi16(t8, rBuffer);
3484
3.76M
            x3 = _mm_unpackhi_epi16(t8, rBuffer);
3485
3486
3.76M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 4));
3487
3.76M
            rBuffer = _mm_mulhi_epi16(x4, r1);
3488
3.76M
            t8 = _mm_mullo_epi16(x5, r0);
3489
3.76M
            t4 = _mm_unpacklo_epi16(t7, rBuffer);
3490
3.76M
            x4 = _mm_unpackhi_epi16(t7, rBuffer);
3491
3492
3.76M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 5));
3493
3.76M
            rBuffer = _mm_mulhi_epi16(x5, r0);
3494
3.76M
            t7 = _mm_mullo_epi16(x6, r1);
3495
3.76M
            t5 = _mm_unpacklo_epi16(t8, rBuffer);
3496
3.76M
            x5 = _mm_unpackhi_epi16(t8, rBuffer);
3497
3498
3.76M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 6));
3499
3.76M
            rBuffer = _mm_mulhi_epi16(x6, r1);
3500
3.76M
            t8 = _mm_mullo_epi16(x7, r0);
3501
3.76M
            t6 = _mm_unpacklo_epi16(t7, rBuffer);
3502
3.76M
            x6 = _mm_unpackhi_epi16(t7, rBuffer);
3503
3504
3.76M
            rBuffer = _mm_mulhi_epi16(x7, r0);
3505
3.76M
            t7 = _mm_unpacklo_epi16(t8, rBuffer);
3506
3.76M
            x7 = _mm_unpackhi_epi16(t8, rBuffer);
3507
3508
3509
3510
            /* add calculus by correct value : */
3511
3512
3.76M
            r1 = _mm_add_epi32(x1, x2);
3513
3.76M
            x3 = _mm_add_epi32(x3, x4);
3514
3.76M
            x5 = _mm_add_epi32(x5, x6);
3515
3.76M
            r1 = _mm_add_epi32(r1, x3);
3516
3517
3.76M
            r1 = _mm_add_epi32(r1, x5);
3518
3519
3.76M
            r0 = _mm_add_epi32(t1, t2);
3520
3.76M
            t3 = _mm_add_epi32(t3, t4);
3521
3.76M
            t5 = _mm_add_epi32(t5, t6);
3522
3.76M
            r0 = _mm_add_epi32(r0, t3);
3523
3.76M
            r0 = _mm_add_epi32(r0, t5);
3524
3.76M
            r1 = _mm_add_epi32(r1, x7);
3525
3.76M
            r0 = _mm_add_epi32(r0, t7);
3526
3.76M
            r1 = _mm_srli_epi32(r1, 6);
3527
3.76M
            r0 = _mm_srli_epi32(r0, 6);
3528
3529
3.76M
            r1 = _mm_and_si128(r1,
3530
3.76M
                    _mm_set_epi16(0, -1, 0, -1, 0, -1, 0, -1));
3531
3.76M
            r0 = _mm_and_si128(r0,
3532
3.76M
                    _mm_set_epi16(0, -1, 0, -1, 0, -1, 0, -1));
3533
3.76M
            r0 = _mm_hadd_epi16(r0, r1);
3534
3.76M
            _mm_store_si128((__m128i *) &dst[x], r0);
3535
3536
3.76M
        }
3537
2.37M
        tmp += MAX_PB_SIZE;
3538
2.37M
        dst += dststride;
3539
2.37M
    }
3540
257k
}
3541
void ff_hevc_put_hevc_qpel_h_1_v_2_sse(int16_t *dst, ptrdiff_t dststride,
3542
                                       const uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
3543
137k
        int16_t* mcbuffer) {
3544
137k
    int x, y;
3545
137k
    uint8_t *src = (uint8_t*) _src;
3546
137k
    ptrdiff_t srcstride = _srcstride / sizeof(uint8_t);
3547
137k
    int16_t *tmp = mcbuffer;
3548
137k
    __m128i x1, x2, x3, x4, x5, x6, x7, x8, rBuffer, rTemp, r0, r1;
3549
137k
    __m128i t1, t2, t3, t4, t5, t6, t7, t8;
3550
3551
137k
    src -= qpel_extra_before[2] * srcstride;
3552
137k
    r0 = _mm_set_epi8(0, 1, -5, 17, 58, -10, 4, -1, 0, 1, -5, 17, 58, -10, 4,
3553
137k
            -1);
3554
3555
    /* LOAD src from memory to registers to limit memory bandwidth */
3556
137k
    if (width == 4) {
3557
3558
286k
        for (y = 0; y < height + qpel_extra[2]; y += 2) {
3559
            /* load data in register     */
3560
254k
            x1 = _mm_loadu_si128((__m128i *) &src[-3]);
3561
254k
            src += srcstride;
3562
254k
            t1 = _mm_loadu_si128((__m128i *) &src[-3]);
3563
254k
            x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
3564
254k
            t2 = _mm_unpacklo_epi64(t1, _mm_srli_si128(t1, 1));
3565
254k
            x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
3566
254k
                    _mm_srli_si128(x1, 3));
3567
254k
            t3 = _mm_unpacklo_epi64(_mm_srli_si128(t1, 2),
3568
254k
                    _mm_srli_si128(t1, 3));
3569
3570
            /*  PMADDUBSW then PMADDW     */
3571
254k
            x2 = _mm_maddubs_epi16(x2, r0);
3572
254k
            t2 = _mm_maddubs_epi16(t2, r0);
3573
254k
            x3 = _mm_maddubs_epi16(x3, r0);
3574
254k
            t3 = _mm_maddubs_epi16(t3, r0);
3575
254k
            x2 = _mm_hadd_epi16(x2, x3);
3576
254k
            t2 = _mm_hadd_epi16(t2, t3);
3577
254k
            x2 = _mm_hadd_epi16(x2, _mm_set1_epi16(0));
3578
254k
            t2 = _mm_hadd_epi16(t2, _mm_set1_epi16(0));
3579
254k
            x2 = _mm_srli_epi16(x2, BIT_DEPTH - 8);
3580
254k
            t2 = _mm_srli_epi16(t2, BIT_DEPTH - 8);
3581
            /* give results back            */
3582
254k
            _mm_storel_epi64((__m128i *) &tmp[0], x2);
3583
3584
254k
            tmp += MAX_PB_SIZE;
3585
254k
            _mm_storel_epi64((__m128i *) &tmp[0], t2);
3586
3587
254k
            src += srcstride;
3588
254k
            tmp += MAX_PB_SIZE;
3589
254k
        }
3590
31.2k
    } else
3591
1.94M
        for (y = 0; y < height + qpel_extra[2]; y++) {
3592
4.68M
            for (x = 0; x < width; x += 8) {
3593
                /* load data in register     */
3594
2.85M
                x1 = _mm_loadu_si128((__m128i *) &src[x - 3]);
3595
2.85M
                x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
3596
2.85M
                x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
3597
2.85M
                        _mm_srli_si128(x1, 3));
3598
2.85M
                x4 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 4),
3599
2.85M
                        _mm_srli_si128(x1, 5));
3600
2.85M
                x5 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 6),
3601
2.85M
                        _mm_srli_si128(x1, 7));
3602
3603
                /*  PMADDUBSW then PMADDW     */
3604
2.85M
                x2 = _mm_maddubs_epi16(x2, r0);
3605
2.85M
                x3 = _mm_maddubs_epi16(x3, r0);
3606
2.85M
                x4 = _mm_maddubs_epi16(x4, r0);
3607
2.85M
                x5 = _mm_maddubs_epi16(x5, r0);
3608
2.85M
                x2 = _mm_hadd_epi16(x2, x3);
3609
2.85M
                x4 = _mm_hadd_epi16(x4, x5);
3610
2.85M
                x2 = _mm_hadd_epi16(x2, x4);
3611
2.85M
                x2 = _mm_srli_si128(x2, BIT_DEPTH - 8);
3612
3613
                /* give results back            */
3614
2.85M
                _mm_store_si128((__m128i *) &tmp[x], x2);
3615
3616
2.85M
            }
3617
1.83M
            src += srcstride;
3618
1.83M
            tmp += MAX_PB_SIZE;
3619
1.83M
        }
3620
3621
137k
    tmp = mcbuffer + qpel_extra_before[2] * MAX_PB_SIZE;
3622
137k
    srcstride = MAX_PB_SIZE;
3623
3624
    /* vertical treatment on temp table : tmp contains 16 bit values, so need to use 32 bit  integers
3625
     for register calculations */
3626
137k
    rTemp = _mm_set_epi16(-1, 4, -11, 40, 40, -11, 4, -1);
3627
1.49M
    for (y = 0; y < height; y++) {
3628
3.47M
        for (x = 0; x < width; x += 8) {
3629
3630
2.12M
            x1 = _mm_load_si128((__m128i *) &tmp[x - 3 * srcstride]);
3631
2.12M
            x2 = _mm_load_si128((__m128i *) &tmp[x - 2 * srcstride]);
3632
2.12M
            x3 = _mm_load_si128((__m128i *) &tmp[x - srcstride]);
3633
2.12M
            x4 = _mm_load_si128((__m128i *) &tmp[x]);
3634
2.12M
            x5 = _mm_load_si128((__m128i *) &tmp[x + srcstride]);
3635
2.12M
            x6 = _mm_load_si128((__m128i *) &tmp[x + 2 * srcstride]);
3636
2.12M
            x7 = _mm_load_si128((__m128i *) &tmp[x + 3 * srcstride]);
3637
2.12M
            x8 = _mm_loadu_si128((__m128i *) &tmp[x + 4 * srcstride]);
3638
3639
2.12M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 0));
3640
2.12M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 1));
3641
2.12M
            t8 = _mm_mullo_epi16(x1, r0);
3642
2.12M
            rBuffer = _mm_mulhi_epi16(x1, r0);
3643
2.12M
            t7 = _mm_mullo_epi16(x2, r1);
3644
2.12M
            t1 = _mm_unpacklo_epi16(t8, rBuffer);
3645
2.12M
            x1 = _mm_unpackhi_epi16(t8, rBuffer);
3646
3647
2.12M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 2));
3648
2.12M
            rBuffer = _mm_mulhi_epi16(x2, r1);
3649
2.12M
            t8 = _mm_mullo_epi16(x3, r0);
3650
2.12M
            t2 = _mm_unpacklo_epi16(t7, rBuffer);
3651
2.12M
            x2 = _mm_unpackhi_epi16(t7, rBuffer);
3652
3653
2.12M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 3));
3654
2.12M
            rBuffer = _mm_mulhi_epi16(x3, r0);
3655
2.12M
            t7 = _mm_mullo_epi16(x4, r1);
3656
2.12M
            t3 = _mm_unpacklo_epi16(t8, rBuffer);
3657
2.12M
            x3 = _mm_unpackhi_epi16(t8, rBuffer);
3658
3659
2.12M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 4));
3660
2.12M
            rBuffer = _mm_mulhi_epi16(x4, r1);
3661
2.12M
            t8 = _mm_mullo_epi16(x5, r0);
3662
2.12M
            t4 = _mm_unpacklo_epi16(t7, rBuffer);
3663
2.12M
            x4 = _mm_unpackhi_epi16(t7, rBuffer);
3664
3665
2.12M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 5));
3666
2.12M
            rBuffer = _mm_mulhi_epi16(x5, r0);
3667
2.12M
            t7 = _mm_mullo_epi16(x6, r1);
3668
2.12M
            t5 = _mm_unpacklo_epi16(t8, rBuffer);
3669
2.12M
            x5 = _mm_unpackhi_epi16(t8, rBuffer);
3670
3671
2.12M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 6));
3672
2.12M
            rBuffer = _mm_mulhi_epi16(x6, r1);
3673
2.12M
            t8 = _mm_mullo_epi16(x7, r0);
3674
2.12M
            t6 = _mm_unpacklo_epi16(t7, rBuffer);
3675
2.12M
            x6 = _mm_unpackhi_epi16(t7, rBuffer);
3676
3677
2.12M
            rBuffer = _mm_mulhi_epi16(x7, r0);
3678
2.12M
            t7 = _mm_unpacklo_epi16(t8, rBuffer);
3679
2.12M
            x7 = _mm_unpackhi_epi16(t8, rBuffer);
3680
3681
2.12M
            t8 = _mm_unpacklo_epi16(
3682
2.12M
                    _mm_mullo_epi16(x8,
3683
2.12M
                            _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))),
3684
2.12M
                            _mm_mulhi_epi16(x8,
3685
2.12M
                                    _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))));
3686
2.12M
            x8 = _mm_unpackhi_epi16(
3687
2.12M
                    _mm_mullo_epi16(x8,
3688
2.12M
                            _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))),
3689
2.12M
                            _mm_mulhi_epi16(x8,
3690
2.12M
                                    _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))));
3691
3692
            /* add calculus by correct value : */
3693
3694
2.12M
            r1 = _mm_add_epi32(x1, x2);
3695
2.12M
            x3 = _mm_add_epi32(x3, x4);
3696
2.12M
            x5 = _mm_add_epi32(x5, x6);
3697
2.12M
            r1 = _mm_add_epi32(r1, x3);
3698
2.12M
            x7 = _mm_add_epi32(x7, x8);
3699
2.12M
            r1 = _mm_add_epi32(r1, x5);
3700
3701
2.12M
            r0 = _mm_add_epi32(t1, t2);
3702
2.12M
            t3 = _mm_add_epi32(t3, t4);
3703
2.12M
            t5 = _mm_add_epi32(t5, t6);
3704
2.12M
            r0 = _mm_add_epi32(r0, t3);
3705
2.12M
            t7 = _mm_add_epi32(t7, t8);
3706
2.12M
            r0 = _mm_add_epi32(r0, t5);
3707
2.12M
            r1 = _mm_add_epi32(r1, x7);
3708
2.12M
            r0 = _mm_add_epi32(r0, t7);
3709
2.12M
            r1 = _mm_srli_epi32(r1, 6);
3710
2.12M
            r0 = _mm_srli_epi32(r0, 6);
3711
3712
2.12M
            r1 = _mm_and_si128(r1,
3713
2.12M
                    _mm_set_epi16(0, -1, 0, -1, 0, -1, 0, -1));
3714
2.12M
            r0 = _mm_and_si128(r0,
3715
2.12M
                    _mm_set_epi16(0, -1, 0, -1, 0, -1, 0, -1));
3716
2.12M
            r0 = _mm_hadd_epi16(r0, r1);
3717
2.12M
            _mm_store_si128((__m128i *) &dst[x], r0);
3718
3719
2.12M
        }
3720
1.35M
        tmp += MAX_PB_SIZE;
3721
1.35M
        dst += dststride;
3722
1.35M
    }
3723
137k
}
3724
void ff_hevc_put_hevc_qpel_h_1_v_3_sse(int16_t *dst, ptrdiff_t dststride,
3725
                                       const uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
3726
145k
        int16_t* mcbuffer) {
3727
145k
    int x, y;
3728
145k
    uint8_t *src = (uint8_t*) _src;
3729
145k
    ptrdiff_t srcstride = _srcstride / sizeof(uint8_t);
3730
145k
    int16_t *tmp = mcbuffer;
3731
145k
    __m128i x1, x2, x3, x4, x5, x6, x7, x8, rBuffer, rTemp, r0, r1;
3732
145k
    __m128i t1, t2, t3, t4, t5, t6, t7, t8;
3733
3734
145k
    src -= qpel_extra_before[3] * srcstride;
3735
145k
    r0 = _mm_set_epi8(0, 1, -5, 17, 58, -10, 4, -1, 0, 1, -5, 17, 58, -10, 4,
3736
145k
            -1);
3737
3738
    /* LOAD src from memory to registers to limit memory bandwidth */
3739
145k
    if (width == 4) {
3740
3741
190k
        for (y = 0; y < height + qpel_extra[3]; y += 2) {
3742
            /* load data in register     */
3743
167k
            x1 = _mm_loadu_si128((__m128i *) &src[-3]);
3744
167k
            src += srcstride;
3745
167k
            t1 = _mm_loadu_si128((__m128i *) &src[-3]);
3746
167k
            x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
3747
167k
            t2 = _mm_unpacklo_epi64(t1, _mm_srli_si128(t1, 1));
3748
167k
            x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
3749
167k
                    _mm_srli_si128(x1, 3));
3750
167k
            t3 = _mm_unpacklo_epi64(_mm_srli_si128(t1, 2),
3751
167k
                    _mm_srli_si128(t1, 3));
3752
3753
            /*  PMADDUBSW then PMADDW     */
3754
167k
            x2 = _mm_maddubs_epi16(x2, r0);
3755
167k
            t2 = _mm_maddubs_epi16(t2, r0);
3756
167k
            x3 = _mm_maddubs_epi16(x3, r0);
3757
167k
            t3 = _mm_maddubs_epi16(t3, r0);
3758
167k
            x2 = _mm_hadd_epi16(x2, x3);
3759
167k
            t2 = _mm_hadd_epi16(t2, t3);
3760
167k
            x2 = _mm_hadd_epi16(x2, _mm_set1_epi16(0));
3761
167k
            t2 = _mm_hadd_epi16(t2, _mm_set1_epi16(0));
3762
167k
            x2 = _mm_srli_epi16(x2, BIT_DEPTH - 8);
3763
167k
            t2 = _mm_srli_epi16(t2, BIT_DEPTH - 8);
3764
            /* give results back            */
3765
167k
            _mm_storel_epi64((__m128i *) &tmp[0], x2);
3766
3767
167k
            tmp += MAX_PB_SIZE;
3768
167k
            _mm_storel_epi64((__m128i *) &tmp[0], t2);
3769
3770
167k
            src += srcstride;
3771
167k
            tmp += MAX_PB_SIZE;
3772
167k
        }
3773
23.6k
    } else
3774
2.05M
        for (y = 0; y < height + qpel_extra[3]; y++) {
3775
4.96M
            for (x = 0; x < width; x += 8) {
3776
                /* load data in register     */
3777
3.03M
                x1 = _mm_loadu_si128((__m128i *) &src[x - 3]);
3778
3.03M
                x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
3779
3.03M
                x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
3780
3.03M
                        _mm_srli_si128(x1, 3));
3781
3.03M
                x4 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 4),
3782
3.03M
                        _mm_srli_si128(x1, 5));
3783
3.03M
                x5 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 6),
3784
3.03M
                        _mm_srli_si128(x1, 7));
3785
3786
                /*  PMADDUBSW then PMADDW     */
3787
3.03M
                x2 = _mm_maddubs_epi16(x2, r0);
3788
3.03M
                x3 = _mm_maddubs_epi16(x3, r0);
3789
3.03M
                x4 = _mm_maddubs_epi16(x4, r0);
3790
3.03M
                x5 = _mm_maddubs_epi16(x5, r0);
3791
3.03M
                x2 = _mm_hadd_epi16(x2, x3);
3792
3.03M
                x4 = _mm_hadd_epi16(x4, x5);
3793
3.03M
                x2 = _mm_hadd_epi16(x2, x4);
3794
3.03M
                x2 = _mm_srli_si128(x2, BIT_DEPTH - 8);
3795
3796
                /* give results back            */
3797
3.03M
                _mm_store_si128((__m128i *) &tmp[x], x2);
3798
3799
3.03M
            }
3800
1.93M
            src += srcstride;
3801
1.93M
            tmp += MAX_PB_SIZE;
3802
1.93M
        }
3803
3804
145k
    tmp = mcbuffer + qpel_extra_before[3] * MAX_PB_SIZE;
3805
145k
    srcstride = MAX_PB_SIZE;
3806
3807
    /* vertical treatment on temp table : tmp contains 16 bit values, so need to use 32 bit  integers
3808
     for register calculations */
3809
145k
    rTemp = _mm_set_epi16(-1, 4, -10, 58, 17, -5, 1, 0);
3810
1.53M
    for (y = 0; y < height; y++) {
3811
3.64M
        for (x = 0; x < width; x += 8) {
3812
3813
2.25M
            x1 = _mm_setzero_si128();
3814
2.25M
            x2 = _mm_load_si128((__m128i *) &tmp[x - 2 * srcstride]);
3815
2.25M
            x3 = _mm_load_si128((__m128i *) &tmp[x - srcstride]);
3816
2.25M
            x4 = _mm_load_si128((__m128i *) &tmp[x]);
3817
2.25M
            x5 = _mm_load_si128((__m128i *) &tmp[x + srcstride]);
3818
2.25M
            x6 = _mm_load_si128((__m128i *) &tmp[x + 2 * srcstride]);
3819
2.25M
            x7 = _mm_load_si128((__m128i *) &tmp[x + 3 * srcstride]);
3820
2.25M
            x8 = _mm_load_si128((__m128i *) &tmp[x + 4 * srcstride]);
3821
3822
3823
2.25M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 1));
3824
3825
2.25M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 2));
3826
2.25M
            t7 = _mm_mullo_epi16(x2, r1);
3827
2.25M
            rBuffer = _mm_mulhi_epi16(x2, r1);
3828
2.25M
            t8 = _mm_mullo_epi16(x3, r0);
3829
2.25M
            t2 = _mm_unpacklo_epi16(t7, rBuffer);
3830
2.25M
            x2 = _mm_unpackhi_epi16(t7, rBuffer);
3831
3832
2.25M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 3));
3833
2.25M
            rBuffer = _mm_mulhi_epi16(x3, r0);
3834
2.25M
            t7 = _mm_mullo_epi16(x4, r1);
3835
2.25M
            t3 = _mm_unpacklo_epi16(t8, rBuffer);
3836
2.25M
            x3 = _mm_unpackhi_epi16(t8, rBuffer);
3837
3838
2.25M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 4));
3839
2.25M
            rBuffer = _mm_mulhi_epi16(x4, r1);
3840
2.25M
            t8 = _mm_mullo_epi16(x5, r0);
3841
2.25M
            t4 = _mm_unpacklo_epi16(t7, rBuffer);
3842
2.25M
            x4 = _mm_unpackhi_epi16(t7, rBuffer);
3843
3844
2.25M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 5));
3845
2.25M
            rBuffer = _mm_mulhi_epi16(x5, r0);
3846
2.25M
            t7 = _mm_mullo_epi16(x6, r1);
3847
2.25M
            t5 = _mm_unpacklo_epi16(t8, rBuffer);
3848
2.25M
            x5 = _mm_unpackhi_epi16(t8, rBuffer);
3849
3850
2.25M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 6));
3851
2.25M
            rBuffer = _mm_mulhi_epi16(x6, r1);
3852
2.25M
            t8 = _mm_mullo_epi16(x7, r0);
3853
2.25M
            t6 = _mm_unpacklo_epi16(t7, rBuffer);
3854
2.25M
            x6 = _mm_unpackhi_epi16(t7, rBuffer);
3855
3856
2.25M
            rBuffer = _mm_mulhi_epi16(x7, r0);
3857
2.25M
            t7 = _mm_unpacklo_epi16(t8, rBuffer);
3858
2.25M
            x7 = _mm_unpackhi_epi16(t8, rBuffer);
3859
3860
2.25M
            t8 = _mm_unpacklo_epi16(
3861
2.25M
                    _mm_mullo_epi16(x8,
3862
2.25M
                            _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))),
3863
2.25M
                            _mm_mulhi_epi16(x8,
3864
2.25M
                                    _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))));
3865
2.25M
            x8 = _mm_unpackhi_epi16(
3866
2.25M
                    _mm_mullo_epi16(x8,
3867
2.25M
                            _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))),
3868
2.25M
                            _mm_mulhi_epi16(x8,
3869
2.25M
                                    _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))));
3870
3871
            /* add calculus by correct value : */
3872
3873
2.25M
            x3 = _mm_add_epi32(x3, x4);
3874
2.25M
            x5 = _mm_add_epi32(x5, x6);
3875
2.25M
            r1 = _mm_add_epi32(x2, x3);
3876
2.25M
            x7 = _mm_add_epi32(x7, x8);
3877
2.25M
            r1 = _mm_add_epi32(r1, x5);
3878
3879
2.25M
            t3 = _mm_add_epi32(t3, t4);
3880
2.25M
            t5 = _mm_add_epi32(t5, t6);
3881
2.25M
            r0 = _mm_add_epi32(t2, t3);
3882
2.25M
            t7 = _mm_add_epi32(t7, t8);
3883
2.25M
            r0 = _mm_add_epi32(r0, t5);
3884
2.25M
            r1 = _mm_add_epi32(r1, x7);
3885
2.25M
            r0 = _mm_add_epi32(r0, t7);
3886
2.25M
            r1 = _mm_srli_epi32(r1, 6);
3887
2.25M
            r0 = _mm_srli_epi32(r0, 6);
3888
3889
2.25M
            r1 = _mm_and_si128(r1,
3890
2.25M
                    _mm_set_epi16(0, -1, 0, -1, 0, -1, 0, -1));
3891
2.25M
            r0 = _mm_and_si128(r0,
3892
2.25M
                    _mm_set_epi16(0, -1, 0, -1, 0, -1, 0, -1));
3893
2.25M
            r0 = _mm_hadd_epi16(r0, r1);
3894
2.25M
            _mm_store_si128((__m128i *) &dst[x], r0);
3895
3896
2.25M
        }
3897
1.39M
        tmp += MAX_PB_SIZE;
3898
1.39M
        dst += dststride;
3899
1.39M
    }
3900
145k
}
3901
void ff_hevc_put_hevc_qpel_h_2_v_1_sse(int16_t *dst, ptrdiff_t dststride,
3902
                                       const uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
3903
190k
        int16_t* mcbuffer) {
3904
190k
    int x, y;
3905
190k
    uint8_t *src = (uint8_t*) _src;
3906
190k
    ptrdiff_t srcstride = _srcstride / sizeof(uint8_t);
3907
190k
    int16_t *tmp = mcbuffer;
3908
190k
    __m128i x1, x2, x3, x4, x5, x6, x7, rBuffer, rTemp, r0, r1;
3909
190k
    __m128i t1, t2, t3, t4, t5, t6, t7, t8;
3910
3911
190k
    src -= qpel_extra_before[1] * srcstride;
3912
190k
    r0 = _mm_set_epi8(-1, 4, -11, 40, 40, -11, 4, -1, -1, 4, -11, 40, 40, -11,
3913
190k
            4, -1);
3914
3915
    /* LOAD src from memory to registers to limit memory bandwidth */
3916
190k
    if (width == 4) {
3917
3918
260k
        for (y = 0; y < height + qpel_extra[1]; y += 2) {
3919
            /* load data in register     */
3920
228k
            x1 = _mm_loadu_si128((__m128i *) &src[-3]);
3921
228k
            src += srcstride;
3922
228k
            t1 = _mm_loadu_si128((__m128i *) &src[-3]);
3923
228k
            x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
3924
228k
            t2 = _mm_unpacklo_epi64(t1, _mm_srli_si128(t1, 1));
3925
228k
            x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
3926
228k
                    _mm_srli_si128(x1, 3));
3927
228k
            t3 = _mm_unpacklo_epi64(_mm_srli_si128(t1, 2),
3928
228k
                    _mm_srli_si128(t1, 3));
3929
3930
            /*  PMADDUBSW then PMADDW     */
3931
228k
            x2 = _mm_maddubs_epi16(x2, r0);
3932
228k
            t2 = _mm_maddubs_epi16(t2, r0);
3933
228k
            x3 = _mm_maddubs_epi16(x3, r0);
3934
228k
            t3 = _mm_maddubs_epi16(t3, r0);
3935
228k
            x2 = _mm_hadd_epi16(x2, x3);
3936
228k
            t2 = _mm_hadd_epi16(t2, t3);
3937
228k
            x2 = _mm_hadd_epi16(x2, _mm_set1_epi16(0));
3938
228k
            t2 = _mm_hadd_epi16(t2, _mm_set1_epi16(0));
3939
228k
            x2 = _mm_srli_epi16(x2, BIT_DEPTH - 8);
3940
228k
            t2 = _mm_srli_epi16(t2, BIT_DEPTH - 8);
3941
            /* give results back            */
3942
228k
            _mm_storel_epi64((__m128i *) &tmp[0], x2);
3943
3944
228k
            tmp += MAX_PB_SIZE;
3945
228k
            _mm_storel_epi64((__m128i *) &tmp[0], t2);
3946
3947
228k
            src += srcstride;
3948
228k
            tmp += MAX_PB_SIZE;
3949
228k
        }
3950
32.3k
    } else
3951
2.66M
        for (y = 0; y < height + qpel_extra[1]; y++) {
3952
6.28M
            for (x = 0; x < width; x += 8) {
3953
                /* load data in register     */
3954
3.78M
                x1 = _mm_loadu_si128((__m128i *) &src[x - 3]);
3955
3.78M
                x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
3956
3.78M
                x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
3957
3.78M
                        _mm_srli_si128(x1, 3));
3958
3.78M
                x4 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 4),
3959
3.78M
                        _mm_srli_si128(x1, 5));
3960
3.78M
                x5 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 6),
3961
3.78M
                        _mm_srli_si128(x1, 7));
3962
3963
                /*  PMADDUBSW then PMADDW     */
3964
3.78M
                x2 = _mm_maddubs_epi16(x2, r0);
3965
3.78M
                x3 = _mm_maddubs_epi16(x3, r0);
3966
3.78M
                x4 = _mm_maddubs_epi16(x4, r0);
3967
3.78M
                x5 = _mm_maddubs_epi16(x5, r0);
3968
3.78M
                x2 = _mm_hadd_epi16(x2, x3);
3969
3.78M
                x4 = _mm_hadd_epi16(x4, x5);
3970
3.78M
                x2 = _mm_hadd_epi16(x2, x4);
3971
3.78M
                x2 = _mm_srli_si128(x2, BIT_DEPTH - 8);
3972
3973
                /* give results back            */
3974
3.78M
                _mm_store_si128((__m128i *) &tmp[x], x2);
3975
3976
3.78M
            }
3977
2.50M
            src += srcstride;
3978
2.50M
            tmp += MAX_PB_SIZE;
3979
2.50M
        }
3980
3981
190k
    tmp = mcbuffer + qpel_extra_before[1] * MAX_PB_SIZE;
3982
190k
    srcstride = MAX_PB_SIZE;
3983
3984
    /* vertical treatment on temp table : tmp contains 16 bit values, so need to use 32 bit  integers
3985
     for register calculations */
3986
190k
    rTemp = _mm_set_epi16(0, 1, -5, 17, 58, -10, 4, -1);
3987
2.00M
    for (y = 0; y < height; y++) {
3988
4.67M
        for (x = 0; x < width; x += 8) {
3989
3990
2.85M
            x1 = _mm_load_si128((__m128i *) &tmp[x - 3 * srcstride]);
3991
2.85M
            x2 = _mm_load_si128((__m128i *) &tmp[x - 2 * srcstride]);
3992
2.85M
            x3 = _mm_load_si128((__m128i *) &tmp[x - srcstride]);
3993
2.85M
            x4 = _mm_load_si128((__m128i *) &tmp[x]);
3994
2.85M
            x5 = _mm_load_si128((__m128i *) &tmp[x + srcstride]);
3995
2.85M
            x6 = _mm_load_si128((__m128i *) &tmp[x + 2 * srcstride]);
3996
2.85M
            x7 = _mm_load_si128((__m128i *) &tmp[x + 3 * srcstride]);
3997
3998
2.85M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 0));
3999
2.85M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 1));
4000
2.85M
            t8 = _mm_mullo_epi16(x1, r0);
4001
2.85M
            rBuffer = _mm_mulhi_epi16(x1, r0);
4002
2.85M
            t7 = _mm_mullo_epi16(x2, r1);
4003
2.85M
            t1 = _mm_unpacklo_epi16(t8, rBuffer);
4004
2.85M
            x1 = _mm_unpackhi_epi16(t8, rBuffer);
4005
4006
2.85M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 2));
4007
2.85M
            rBuffer = _mm_mulhi_epi16(x2, r1);
4008
2.85M
            t8 = _mm_mullo_epi16(x3, r0);
4009
2.85M
            t2 = _mm_unpacklo_epi16(t7, rBuffer);
4010
2.85M
            x2 = _mm_unpackhi_epi16(t7, rBuffer);
4011
4012
2.85M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 3));
4013
2.85M
            rBuffer = _mm_mulhi_epi16(x3, r0);
4014
2.85M
            t7 = _mm_mullo_epi16(x4, r1);
4015
2.85M
            t3 = _mm_unpacklo_epi16(t8, rBuffer);
4016
2.85M
            x3 = _mm_unpackhi_epi16(t8, rBuffer);
4017
4018
2.85M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 4));
4019
2.85M
            rBuffer = _mm_mulhi_epi16(x4, r1);
4020
2.85M
            t8 = _mm_mullo_epi16(x5, r0);
4021
2.85M
            t4 = _mm_unpacklo_epi16(t7, rBuffer);
4022
2.85M
            x4 = _mm_unpackhi_epi16(t7, rBuffer);
4023
4024
2.85M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 5));
4025
2.85M
            rBuffer = _mm_mulhi_epi16(x5, r0);
4026
2.85M
            t7 = _mm_mullo_epi16(x6, r1);
4027
2.85M
            t5 = _mm_unpacklo_epi16(t8, rBuffer);
4028
2.85M
            x5 = _mm_unpackhi_epi16(t8, rBuffer);
4029
4030
2.85M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 6));
4031
2.85M
            rBuffer = _mm_mulhi_epi16(x6, r1);
4032
2.85M
            t8 = _mm_mullo_epi16(x7, r0);
4033
2.85M
            t6 = _mm_unpacklo_epi16(t7, rBuffer);
4034
2.85M
            x6 = _mm_unpackhi_epi16(t7, rBuffer);
4035
4036
2.85M
            rBuffer = _mm_mulhi_epi16(x7, r0);
4037
2.85M
            t7 = _mm_unpacklo_epi16(t8, rBuffer);
4038
2.85M
            x7 = _mm_unpackhi_epi16(t8, rBuffer);
4039
4040
4041
4042
            /* add calculus by correct value : */
4043
4044
2.85M
            r1 = _mm_add_epi32(x1, x2);
4045
2.85M
            x3 = _mm_add_epi32(x3, x4);
4046
2.85M
            x5 = _mm_add_epi32(x5, x6);
4047
2.85M
            r1 = _mm_add_epi32(r1, x3);
4048
2.85M
            r1 = _mm_add_epi32(r1, x5);
4049
4050
2.85M
            r0 = _mm_add_epi32(t1, t2);
4051
2.85M
            t3 = _mm_add_epi32(t3, t4);
4052
2.85M
            t5 = _mm_add_epi32(t5, t6);
4053
2.85M
            r0 = _mm_add_epi32(r0, t3);
4054
2.85M
            r0 = _mm_add_epi32(r0, t5);
4055
2.85M
            r1 = _mm_add_epi32(r1, x7);
4056
2.85M
            r0 = _mm_add_epi32(r0, t7);
4057
2.85M
            r1 = _mm_srli_epi32(r1, 6);
4058
2.85M
            r0 = _mm_srli_epi32(r0, 6);
4059
4060
2.85M
            r1 = _mm_and_si128(r1,
4061
2.85M
                    _mm_set_epi16(0, -1, 0, -1, 0, -1, 0, -1));
4062
2.85M
            r0 = _mm_and_si128(r0,
4063
2.85M
                    _mm_set_epi16(0, -1, 0, -1, 0, -1, 0, -1));
4064
2.85M
            r0 = _mm_hadd_epi16(r0, r1);
4065
2.85M
            _mm_store_si128((__m128i *) &dst[x], r0);
4066
4067
2.85M
        }
4068
1.81M
        tmp += MAX_PB_SIZE;
4069
1.81M
        dst += dststride;
4070
1.81M
    }
4071
190k
}
4072
void ff_hevc_put_hevc_qpel_h_2_v_2_sse(int16_t *dst, ptrdiff_t dststride,
4073
                                       const uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
4074
295k
        int16_t* mcbuffer) {
4075
295k
    int x, y;
4076
295k
    uint8_t *src = (uint8_t*) _src;
4077
295k
    ptrdiff_t srcstride = _srcstride / sizeof(uint8_t);
4078
295k
    int16_t *tmp = mcbuffer;
4079
295k
    __m128i x1, x2, x3, x4, x5, x6, x7, x8, rBuffer, rTemp, r0, r1;
4080
295k
    __m128i t1, t2, t3, t4, t5, t6, t7, t8;
4081
4082
295k
    src -= qpel_extra_before[2] * srcstride;
4083
295k
    r0 = _mm_set_epi8(-1, 4, -11, 40, 40, -11, 4, -1, -1, 4, -11, 40, 40, -11,
4084
295k
            4, -1);
4085
4086
    /* LOAD src from memory to registers to limit memory bandwidth */
4087
295k
    if (width == 4) {
4088
4089
546k
        for (y = 0; y < height + qpel_extra[2]; y += 2) {
4090
            /* load data in register     */
4091
486k
            x1 = _mm_loadu_si128((__m128i *) &src[-3]);
4092
486k
            src += srcstride;
4093
486k
            t1 = _mm_loadu_si128((__m128i *) &src[-3]);
4094
486k
            x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
4095
486k
            t2 = _mm_unpacklo_epi64(t1, _mm_srli_si128(t1, 1));
4096
486k
            x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
4097
486k
                    _mm_srli_si128(x1, 3));
4098
486k
            t3 = _mm_unpacklo_epi64(_mm_srli_si128(t1, 2),
4099
486k
                    _mm_srli_si128(t1, 3));
4100
4101
            /*  PMADDUBSW then PMADDW     */
4102
486k
            x2 = _mm_maddubs_epi16(x2, r0);
4103
486k
            t2 = _mm_maddubs_epi16(t2, r0);
4104
486k
            x3 = _mm_maddubs_epi16(x3, r0);
4105
486k
            t3 = _mm_maddubs_epi16(t3, r0);
4106
486k
            x2 = _mm_hadd_epi16(x2, x3);
4107
486k
            t2 = _mm_hadd_epi16(t2, t3);
4108
486k
            x2 = _mm_hadd_epi16(x2, _mm_set1_epi16(0));
4109
486k
            t2 = _mm_hadd_epi16(t2, _mm_set1_epi16(0));
4110
486k
            x2 = _mm_srli_epi16(x2, BIT_DEPTH - 8);
4111
486k
            t2 = _mm_srli_epi16(t2, BIT_DEPTH - 8);
4112
            /* give results back            */
4113
486k
            _mm_storel_epi64((__m128i *) &tmp[0], x2);
4114
4115
486k
            tmp += MAX_PB_SIZE;
4116
486k
            _mm_storel_epi64((__m128i *) &tmp[0], t2);
4117
4118
486k
            src += srcstride;
4119
486k
            tmp += MAX_PB_SIZE;
4120
486k
        }
4121
60.3k
    } else
4122
3.92M
        for (y = 0; y < height + qpel_extra[2]; y++) {
4123
8.75M
            for (x = 0; x < width; x += 8) {
4124
                /* load data in register     */
4125
5.05M
                x1 = _mm_loadu_si128((__m128i *) &src[x - 3]);
4126
5.05M
                x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
4127
5.05M
                x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
4128
5.05M
                        _mm_srli_si128(x1, 3));
4129
5.05M
                x4 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 4),
4130
5.05M
                        _mm_srli_si128(x1, 5));
4131
5.05M
                x5 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 6),
4132
5.05M
                        _mm_srli_si128(x1, 7));
4133
4134
                /*  PMADDUBSW then PMADDW     */
4135
5.05M
                x2 = _mm_maddubs_epi16(x2, r0);
4136
5.05M
                x3 = _mm_maddubs_epi16(x3, r0);
4137
5.05M
                x4 = _mm_maddubs_epi16(x4, r0);
4138
5.05M
                x5 = _mm_maddubs_epi16(x5, r0);
4139
5.05M
                x2 = _mm_hadd_epi16(x2, x3);
4140
5.05M
                x4 = _mm_hadd_epi16(x4, x5);
4141
5.05M
                x2 = _mm_hadd_epi16(x2, x4);
4142
5.05M
                x2 = _mm_srli_si128(x2, BIT_DEPTH - 8);
4143
4144
                /* give results back            */
4145
5.05M
                _mm_store_si128((__m128i *) &tmp[x], x2);
4146
4147
5.05M
            }
4148
3.69M
            src += srcstride;
4149
3.69M
            tmp += MAX_PB_SIZE;
4150
3.69M
        }
4151
4152
295k
    tmp = mcbuffer + qpel_extra_before[2] * MAX_PB_SIZE;
4153
295k
    srcstride = MAX_PB_SIZE;
4154
4155
    /* vertical treatment on temp table : tmp contains 16 bit values, so need to use 32 bit  integers
4156
     for register calculations */
4157
295k
    rTemp = _mm_set_epi16(-1, 4, -11, 40, 40, -11, 4, -1);
4158
2.83M
    for (y = 0; y < height; y++) {
4159
6.11M
        for (x = 0; x < width; x += 8) {
4160
4161
3.57M
            x1 = _mm_load_si128((__m128i *) &tmp[x - 3 * srcstride]);
4162
3.57M
            x2 = _mm_load_si128((__m128i *) &tmp[x - 2 * srcstride]);
4163
3.57M
            x3 = _mm_load_si128((__m128i *) &tmp[x - srcstride]);
4164
3.57M
            x4 = _mm_load_si128((__m128i *) &tmp[x]);
4165
3.57M
            x5 = _mm_load_si128((__m128i *) &tmp[x + srcstride]);
4166
3.57M
            x6 = _mm_load_si128((__m128i *) &tmp[x + 2 * srcstride]);
4167
3.57M
            x7 = _mm_load_si128((__m128i *) &tmp[x + 3 * srcstride]);
4168
3.57M
            x8 = _mm_load_si128((__m128i *) &tmp[x + 4 * srcstride]);
4169
4170
3.57M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 0));
4171
3.57M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 1));
4172
3.57M
            t8 = _mm_mullo_epi16(x1, r0);
4173
3.57M
            rBuffer = _mm_mulhi_epi16(x1, r0);
4174
3.57M
            t7 = _mm_mullo_epi16(x2, r1);
4175
3.57M
            t1 = _mm_unpacklo_epi16(t8, rBuffer);
4176
3.57M
            x1 = _mm_unpackhi_epi16(t8, rBuffer);
4177
4178
3.57M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 2));
4179
3.57M
            rBuffer = _mm_mulhi_epi16(x2, r1);
4180
3.57M
            t8 = _mm_mullo_epi16(x3, r0);
4181
3.57M
            t2 = _mm_unpacklo_epi16(t7, rBuffer);
4182
3.57M
            x2 = _mm_unpackhi_epi16(t7, rBuffer);
4183
4184
3.57M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 3));
4185
3.57M
            rBuffer = _mm_mulhi_epi16(x3, r0);
4186
3.57M
            t7 = _mm_mullo_epi16(x4, r1);
4187
3.57M
            t3 = _mm_unpacklo_epi16(t8, rBuffer);
4188
3.57M
            x3 = _mm_unpackhi_epi16(t8, rBuffer);
4189
4190
3.57M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 4));
4191
3.57M
            rBuffer = _mm_mulhi_epi16(x4, r1);
4192
3.57M
            t8 = _mm_mullo_epi16(x5, r0);
4193
3.57M
            t4 = _mm_unpacklo_epi16(t7, rBuffer);
4194
3.57M
            x4 = _mm_unpackhi_epi16(t7, rBuffer);
4195
4196
3.57M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 5));
4197
3.57M
            rBuffer = _mm_mulhi_epi16(x5, r0);
4198
3.57M
            t7 = _mm_mullo_epi16(x6, r1);
4199
3.57M
            t5 = _mm_unpacklo_epi16(t8, rBuffer);
4200
3.57M
            x5 = _mm_unpackhi_epi16(t8, rBuffer);
4201
4202
3.57M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 6));
4203
3.57M
            rBuffer = _mm_mulhi_epi16(x6, r1);
4204
3.57M
            t8 = _mm_mullo_epi16(x7, r0);
4205
3.57M
            t6 = _mm_unpacklo_epi16(t7, rBuffer);
4206
3.57M
            x6 = _mm_unpackhi_epi16(t7, rBuffer);
4207
4208
3.57M
            rBuffer = _mm_mulhi_epi16(x7, r0);
4209
3.57M
            t7 = _mm_unpacklo_epi16(t8, rBuffer);
4210
3.57M
            x7 = _mm_unpackhi_epi16(t8, rBuffer);
4211
4212
3.57M
            t8 = _mm_unpacklo_epi16(
4213
3.57M
                    _mm_mullo_epi16(x8,
4214
3.57M
                            _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))),
4215
3.57M
                            _mm_mulhi_epi16(x8,
4216
3.57M
                                    _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))));
4217
3.57M
            x8 = _mm_unpackhi_epi16(
4218
3.57M
                    _mm_mullo_epi16(x8,
4219
3.57M
                            _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))),
4220
3.57M
                            _mm_mulhi_epi16(x8,
4221
3.57M
                                    _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))));
4222
4223
            /* add calculus by correct value : */
4224
4225
3.57M
            r1 = _mm_add_epi32(x1, x2);
4226
3.57M
            x3 = _mm_add_epi32(x3, x4);
4227
3.57M
            x5 = _mm_add_epi32(x5, x6);
4228
3.57M
            r1 = _mm_add_epi32(r1, x3);
4229
3.57M
            x7 = _mm_add_epi32(x7, x8);
4230
3.57M
            r1 = _mm_add_epi32(r1, x5);
4231
4232
3.57M
            r0 = _mm_add_epi32(t1, t2);
4233
3.57M
            t3 = _mm_add_epi32(t3, t4);
4234
3.57M
            t5 = _mm_add_epi32(t5, t6);
4235
3.57M
            r0 = _mm_add_epi32(r0, t3);
4236
3.57M
            t7 = _mm_add_epi32(t7, t8);
4237
3.57M
            r0 = _mm_add_epi32(r0, t5);
4238
3.57M
            r1 = _mm_add_epi32(r1, x7);
4239
3.57M
            r0 = _mm_add_epi32(r0, t7);
4240
3.57M
            r1 = _mm_srli_epi32(r1, 6);
4241
3.57M
            r0 = _mm_srli_epi32(r0, 6);
4242
4243
3.57M
            r1 = _mm_and_si128(r1,
4244
3.57M
                    _mm_set_epi16(0, -1, 0, -1, 0, -1, 0, -1));
4245
3.57M
            r0 = _mm_and_si128(r0,
4246
3.57M
                    _mm_set_epi16(0, -1, 0, -1, 0, -1, 0, -1));
4247
3.57M
            r0 = _mm_hadd_epi16(r0, r1);
4248
3.57M
            _mm_store_si128((__m128i *) &dst[x], r0);
4249
4250
3.57M
        }
4251
2.54M
        tmp += MAX_PB_SIZE;
4252
2.54M
        dst += dststride;
4253
2.54M
    }
4254
295k
}
4255
void ff_hevc_put_hevc_qpel_h_2_v_3_sse(int16_t *dst, ptrdiff_t dststride,
4256
                                       const uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
4257
111k
        int16_t* mcbuffer) {
4258
111k
    int x, y;
4259
111k
    uint8_t *src = (uint8_t*) _src;
4260
111k
    ptrdiff_t srcstride = _srcstride / sizeof(uint8_t);
4261
111k
    int16_t *tmp = mcbuffer;
4262
111k
    __m128i x1, x2, x3, x4, x5, x6, x7, x8, rBuffer, rTemp, r0, r1;
4263
111k
    __m128i t1, t2, t3, t4, t5, t6, t7, t8;
4264
4265
111k
    src -= qpel_extra_before[3] * srcstride;
4266
111k
    r0 = _mm_set_epi8(-1, 4, -11, 40, 40, -11, 4, -1, -1, 4, -11, 40, 40, -11,
4267
111k
            4, -1);
4268
4269
    /* LOAD src from memory to registers to limit memory bandwidth */
4270
111k
    if (width == 4) {
4271
4272
189k
        for (y = 0; y < height + qpel_extra[3]; y += 2) {
4273
            /* load data in register     */
4274
166k
            x1 = _mm_loadu_si128((__m128i *) &src[-3]);
4275
166k
            src += srcstride;
4276
166k
            t1 = _mm_loadu_si128((__m128i *) &src[-3]);
4277
166k
            x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
4278
166k
            t2 = _mm_unpacklo_epi64(t1, _mm_srli_si128(t1, 1));
4279
166k
            x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
4280
166k
                    _mm_srli_si128(x1, 3));
4281
166k
            t3 = _mm_unpacklo_epi64(_mm_srli_si128(t1, 2),
4282
166k
                    _mm_srli_si128(t1, 3));
4283
4284
            /*  PMADDUBSW then PMADDW     */
4285
166k
            x2 = _mm_maddubs_epi16(x2, r0);
4286
166k
            t2 = _mm_maddubs_epi16(t2, r0);
4287
166k
            x3 = _mm_maddubs_epi16(x3, r0);
4288
166k
            t3 = _mm_maddubs_epi16(t3, r0);
4289
166k
            x2 = _mm_hadd_epi16(x2, x3);
4290
166k
            t2 = _mm_hadd_epi16(t2, t3);
4291
166k
            x2 = _mm_hadd_epi16(x2, _mm_set1_epi16(0));
4292
166k
            t2 = _mm_hadd_epi16(t2, _mm_set1_epi16(0));
4293
166k
            x2 = _mm_srli_epi16(x2, BIT_DEPTH - 8);
4294
166k
            t2 = _mm_srli_epi16(t2, BIT_DEPTH - 8);
4295
            /* give results back            */
4296
166k
            _mm_storel_epi64((__m128i *) &tmp[0], x2);
4297
4298
166k
            tmp += MAX_PB_SIZE;
4299
166k
            _mm_storel_epi64((__m128i *) &tmp[0], t2);
4300
4301
166k
            src += srcstride;
4302
166k
            tmp += MAX_PB_SIZE;
4303
166k
        }
4304
23.4k
    } else
4305
1.51M
        for (y = 0; y < height + qpel_extra[3]; y++) {
4306
3.59M
            for (x = 0; x < width; x += 8) {
4307
                /* load data in register     */
4308
2.17M
                x1 = _mm_loadu_si128((__m128i *) &src[x - 3]);
4309
2.17M
                x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
4310
2.17M
                x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
4311
2.17M
                        _mm_srli_si128(x1, 3));
4312
2.17M
                x4 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 4),
4313
2.17M
                        _mm_srli_si128(x1, 5));
4314
2.17M
                x5 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 6),
4315
2.17M
                        _mm_srli_si128(x1, 7));
4316
4317
                /*  PMADDUBSW then PMADDW     */
4318
2.17M
                x2 = _mm_maddubs_epi16(x2, r0);
4319
2.17M
                x3 = _mm_maddubs_epi16(x3, r0);
4320
2.17M
                x4 = _mm_maddubs_epi16(x4, r0);
4321
2.17M
                x5 = _mm_maddubs_epi16(x5, r0);
4322
2.17M
                x2 = _mm_hadd_epi16(x2, x3);
4323
2.17M
                x4 = _mm_hadd_epi16(x4, x5);
4324
2.17M
                x2 = _mm_hadd_epi16(x2, x4);
4325
2.17M
                x2 = _mm_srli_si128(x2, BIT_DEPTH - 8);
4326
4327
                /* give results back            */
4328
2.17M
                _mm_store_si128((__m128i *) &tmp[x], x2);
4329
4330
2.17M
            }
4331
1.42M
            src += srcstride;
4332
1.42M
            tmp += MAX_PB_SIZE;
4333
1.42M
        }
4334
4335
111k
    tmp = mcbuffer + qpel_extra_before[3] * MAX_PB_SIZE;
4336
111k
    srcstride = MAX_PB_SIZE;
4337
4338
    /* vertical treatment on temp table : tmp contains 16 bit values, so need to use 32 bit  integers
4339
     for register calculations */
4340
111k
    rTemp = _mm_set_epi16(-1, 4, -10, 58, 17, -5, 1, 0);
4341
1.20M
    for (y = 0; y < height; y++) {
4342
2.76M
        for (x = 0; x < width; x += 8) {
4343
4344
1.67M
            x1 = _mm_setzero_si128();
4345
1.67M
            x2 = _mm_load_si128((__m128i *) &tmp[x - 2 * srcstride]);
4346
1.67M
            x3 = _mm_load_si128((__m128i *) &tmp[x - srcstride]);
4347
1.67M
            x4 = _mm_load_si128((__m128i *) &tmp[x]);
4348
1.67M
            x5 = _mm_load_si128((__m128i *) &tmp[x + srcstride]);
4349
1.67M
            x6 = _mm_load_si128((__m128i *) &tmp[x + 2 * srcstride]);
4350
1.67M
            x7 = _mm_load_si128((__m128i *) &tmp[x + 3 * srcstride]);
4351
1.67M
            x8 = _mm_load_si128((__m128i *) &tmp[x + 4 * srcstride]);
4352
4353
1.67M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 1));
4354
4355
1.67M
            t7 = _mm_mullo_epi16(x2, r1);
4356
4357
4358
1.67M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 2));
4359
1.67M
            rBuffer = _mm_mulhi_epi16(x2, r1);
4360
1.67M
            t8 = _mm_mullo_epi16(x3, r0);
4361
1.67M
            t2 = _mm_unpacklo_epi16(t7, rBuffer);
4362
1.67M
            x2 = _mm_unpackhi_epi16(t7, rBuffer);
4363
4364
1.67M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 3));
4365
1.67M
            rBuffer = _mm_mulhi_epi16(x3, r0);
4366
1.67M
            t7 = _mm_mullo_epi16(x4, r1);
4367
1.67M
            t3 = _mm_unpacklo_epi16(t8, rBuffer);
4368
1.67M
            x3 = _mm_unpackhi_epi16(t8, rBuffer);
4369
4370
1.67M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 4));
4371
1.67M
            rBuffer = _mm_mulhi_epi16(x4, r1);
4372
1.67M
            t8 = _mm_mullo_epi16(x5, r0);
4373
1.67M
            t4 = _mm_unpacklo_epi16(t7, rBuffer);
4374
1.67M
            x4 = _mm_unpackhi_epi16(t7, rBuffer);
4375
4376
1.67M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 5));
4377
1.67M
            rBuffer = _mm_mulhi_epi16(x5, r0);
4378
1.67M
            t7 = _mm_mullo_epi16(x6, r1);
4379
1.67M
            t5 = _mm_unpacklo_epi16(t8, rBuffer);
4380
1.67M
            x5 = _mm_unpackhi_epi16(t8, rBuffer);
4381
4382
1.67M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 6));
4383
1.67M
            rBuffer = _mm_mulhi_epi16(x6, r1);
4384
1.67M
            t8 = _mm_mullo_epi16(x7, r0);
4385
1.67M
            t6 = _mm_unpacklo_epi16(t7, rBuffer);
4386
1.67M
            x6 = _mm_unpackhi_epi16(t7, rBuffer);
4387
4388
1.67M
            rBuffer = _mm_mulhi_epi16(x7, r0);
4389
1.67M
            t7 = _mm_unpacklo_epi16(t8, rBuffer);
4390
1.67M
            x7 = _mm_unpackhi_epi16(t8, rBuffer);
4391
4392
1.67M
            t8 = _mm_unpacklo_epi16(
4393
1.67M
                    _mm_mullo_epi16(x8,
4394
1.67M
                            _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))),
4395
1.67M
                            _mm_mulhi_epi16(x8,
4396
1.67M
                                    _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))));
4397
1.67M
            x8 = _mm_unpackhi_epi16(
4398
1.67M
                    _mm_mullo_epi16(x8,
4399
1.67M
                            _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))),
4400
1.67M
                            _mm_mulhi_epi16(x8,
4401
1.67M
                                    _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))));
4402
4403
            /* add calculus by correct value : */
4404
4405
1.67M
            x3 = _mm_add_epi32(x3, x4);
4406
1.67M
            x5 = _mm_add_epi32(x5, x6);
4407
1.67M
            r1 = _mm_add_epi32(x2, x3);
4408
1.67M
            x7 = _mm_add_epi32(x7, x8);
4409
1.67M
            r1 = _mm_add_epi32(r1, x5);
4410
4411
1.67M
            t3 = _mm_add_epi32(t3, t4);
4412
1.67M
            t5 = _mm_add_epi32(t5, t6);
4413
1.67M
            r0 = _mm_add_epi32(t2, t3);
4414
1.67M
            t7 = _mm_add_epi32(t7, t8);
4415
1.67M
            r0 = _mm_add_epi32(r0, t5);
4416
1.67M
            r1 = _mm_add_epi32(r1, x7);
4417
1.67M
            r0 = _mm_add_epi32(r0, t7);
4418
1.67M
            r1 = _mm_srli_epi32(r1, 6);
4419
1.67M
            r0 = _mm_srli_epi32(r0, 6);
4420
4421
1.67M
            r1 = _mm_and_si128(r1,
4422
1.67M
                    _mm_set_epi16(0, -1, 0, -1, 0, -1, 0, -1));
4423
1.67M
            r0 = _mm_and_si128(r0,
4424
1.67M
                    _mm_set_epi16(0, -1, 0, -1, 0, -1, 0, -1));
4425
1.67M
            r0 = _mm_hadd_epi16(r0, r1);
4426
1.67M
            _mm_store_si128((__m128i *) &dst[x], r0);
4427
4428
1.67M
        }
4429
1.09M
        tmp += MAX_PB_SIZE;
4430
1.09M
        dst += dststride;
4431
1.09M
    }
4432
111k
}
4433
void ff_hevc_put_hevc_qpel_h_3_v_1_sse(int16_t *dst, ptrdiff_t dststride,
4434
                                       const uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
4435
149k
        int16_t* mcbuffer) {
4436
149k
    int x, y;
4437
149k
    uint8_t *src = (uint8_t*) _src;
4438
149k
    ptrdiff_t srcstride = _srcstride / sizeof(uint8_t);
4439
149k
    int16_t *tmp = mcbuffer;
4440
149k
    __m128i x1, x2, x3, x4, x5, x6, x7, rBuffer, rTemp, r0, r1;
4441
149k
    __m128i t1, t2, t3, t4, t5, t6, t7, t8;
4442
4443
149k
    src -= qpel_extra_before[1] * srcstride;
4444
149k
    r0 = _mm_set_epi8(-1, 4, -10, 58, 17, -5, 1, 0, -1, 4, -10, 58, 17, -5, 1,
4445
149k
            0);
4446
4447
    /* LOAD src from memory to registers to limit memory bandwidth */
4448
149k
    if (width == 4) {
4449
4450
177k
        for (y = 0; y < height + qpel_extra[1]; y += 2) {
4451
            /* load data in register     */
4452
155k
            x1 = _mm_loadu_si128((__m128i *) &src[-2]);
4453
155k
            x1 = _mm_slli_si128(x1, 1);
4454
155k
            src += srcstride;
4455
155k
            t1 = _mm_loadu_si128((__m128i *) &src[-2]);
4456
155k
            t1 = _mm_slli_si128(t1, 1);
4457
155k
            x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
4458
155k
            t2 = _mm_unpacklo_epi64(t1, _mm_srli_si128(t1, 1));
4459
155k
            x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
4460
155k
                    _mm_srli_si128(x1, 3));
4461
155k
            t3 = _mm_unpacklo_epi64(_mm_srli_si128(t1, 2),
4462
155k
                    _mm_srli_si128(t1, 3));
4463
4464
            /*  PMADDUBSW then PMADDW     */
4465
155k
            x2 = _mm_maddubs_epi16(x2, r0);
4466
155k
            t2 = _mm_maddubs_epi16(t2, r0);
4467
155k
            x3 = _mm_maddubs_epi16(x3, r0);
4468
155k
            t3 = _mm_maddubs_epi16(t3, r0);
4469
155k
            x2 = _mm_hadd_epi16(x2, x3);
4470
155k
            t2 = _mm_hadd_epi16(t2, t3);
4471
155k
            x2 = _mm_hadd_epi16(x2, _mm_set1_epi16(0));
4472
155k
            t2 = _mm_hadd_epi16(t2, _mm_set1_epi16(0));
4473
155k
            x2 = _mm_srli_epi16(x2, BIT_DEPTH - 8);
4474
155k
            t2 = _mm_srli_epi16(t2, BIT_DEPTH - 8);
4475
            /* give results back            */
4476
155k
            _mm_storel_epi64((__m128i *) &tmp[0], x2);
4477
4478
155k
            tmp += MAX_PB_SIZE;
4479
155k
            _mm_storel_epi64((__m128i *) &tmp[0], t2);
4480
4481
155k
            src += srcstride;
4482
155k
            tmp += MAX_PB_SIZE;
4483
155k
        }
4484
21.8k
    } else
4485
2.15M
        for (y = 0; y < height + qpel_extra[1]; y++) {
4486
5.16M
            for (x = 0; x < width; x += 8) {
4487
                /* load data in register     */
4488
3.13M
                x1 = _mm_loadu_si128((__m128i *) &src[x - 2]);
4489
3.13M
                x1 = _mm_slli_si128(x1, 1);
4490
3.13M
                x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
4491
3.13M
                x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
4492
3.13M
                        _mm_srli_si128(x1, 3));
4493
3.13M
                x4 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 4),
4494
3.13M
                        _mm_srli_si128(x1, 5));
4495
3.13M
                x5 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 6),
4496
3.13M
                        _mm_srli_si128(x1, 7));
4497
4498
                /*  PMADDUBSW then PMADDW     */
4499
3.13M
                x2 = _mm_maddubs_epi16(x2, r0);
4500
3.13M
                x3 = _mm_maddubs_epi16(x3, r0);
4501
3.13M
                x4 = _mm_maddubs_epi16(x4, r0);
4502
3.13M
                x5 = _mm_maddubs_epi16(x5, r0);
4503
3.13M
                x2 = _mm_hadd_epi16(x2, x3);
4504
3.13M
                x4 = _mm_hadd_epi16(x4, x5);
4505
3.13M
                x2 = _mm_hadd_epi16(x2, x4);
4506
3.13M
                x2 = _mm_srli_si128(x2, BIT_DEPTH - 8);
4507
4508
                /* give results back            */
4509
3.13M
                _mm_store_si128((__m128i *) &tmp[x], x2);
4510
4511
3.13M
            }
4512
2.02M
            src += srcstride;
4513
2.02M
            tmp += MAX_PB_SIZE;
4514
2.02M
        }
4515
4516
149k
    tmp = mcbuffer + qpel_extra_before[1] * MAX_PB_SIZE;
4517
149k
    srcstride = MAX_PB_SIZE;
4518
4519
    /* vertical treatment on temp table : tmp contains 16 bit values, so need to use 32 bit  integers
4520
     for register calculations */
4521
149k
    rTemp = _mm_set_epi16(0, 1, -5, 17, 58, -10, 4, -1);
4522
1.59M
    for (y = 0; y < height; y++) {
4523
3.75M
        for (x = 0; x < width; x += 8) {
4524
4525
2.31M
            x1 = _mm_load_si128((__m128i *) &tmp[x - 3 * srcstride]);
4526
2.31M
            x2 = _mm_load_si128((__m128i *) &tmp[x - 2 * srcstride]);
4527
2.31M
            x3 = _mm_load_si128((__m128i *) &tmp[x - srcstride]);
4528
2.31M
            x4 = _mm_load_si128((__m128i *) &tmp[x]);
4529
2.31M
            x5 = _mm_load_si128((__m128i *) &tmp[x + srcstride]);
4530
2.31M
            x6 = _mm_load_si128((__m128i *) &tmp[x + 2 * srcstride]);
4531
2.31M
            x7 = _mm_load_si128((__m128i *) &tmp[x + 3 * srcstride]);
4532
4533
2.31M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 0));
4534
2.31M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 1));
4535
2.31M
            t8 = _mm_mullo_epi16(x1, r0);
4536
2.31M
            rBuffer = _mm_mulhi_epi16(x1, r0);
4537
2.31M
            t7 = _mm_mullo_epi16(x2, r1);
4538
2.31M
            t1 = _mm_unpacklo_epi16(t8, rBuffer);
4539
2.31M
            x1 = _mm_unpackhi_epi16(t8, rBuffer);
4540
4541
2.31M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 2));
4542
2.31M
            rBuffer = _mm_mulhi_epi16(x2, r1);
4543
2.31M
            t8 = _mm_mullo_epi16(x3, r0);
4544
2.31M
            t2 = _mm_unpacklo_epi16(t7, rBuffer);
4545
2.31M
            x2 = _mm_unpackhi_epi16(t7, rBuffer);
4546
4547
2.31M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 3));
4548
2.31M
            rBuffer = _mm_mulhi_epi16(x3, r0);
4549
2.31M
            t7 = _mm_mullo_epi16(x4, r1);
4550
2.31M
            t3 = _mm_unpacklo_epi16(t8, rBuffer);
4551
2.31M
            x3 = _mm_unpackhi_epi16(t8, rBuffer);
4552
4553
2.31M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 4));
4554
2.31M
            rBuffer = _mm_mulhi_epi16(x4, r1);
4555
2.31M
            t8 = _mm_mullo_epi16(x5, r0);
4556
2.31M
            t4 = _mm_unpacklo_epi16(t7, rBuffer);
4557
2.31M
            x4 = _mm_unpackhi_epi16(t7, rBuffer);
4558
4559
2.31M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 5));
4560
2.31M
            rBuffer = _mm_mulhi_epi16(x5, r0);
4561
2.31M
            t7 = _mm_mullo_epi16(x6, r1);
4562
2.31M
            t5 = _mm_unpacklo_epi16(t8, rBuffer);
4563
2.31M
            x5 = _mm_unpackhi_epi16(t8, rBuffer);
4564
4565
2.31M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 6));
4566
2.31M
            rBuffer = _mm_mulhi_epi16(x6, r1);
4567
2.31M
            t8 = _mm_mullo_epi16(x7, r0);
4568
2.31M
            t6 = _mm_unpacklo_epi16(t7, rBuffer);
4569
2.31M
            x6 = _mm_unpackhi_epi16(t7, rBuffer);
4570
4571
2.31M
            rBuffer = _mm_mulhi_epi16(x7, r0);
4572
2.31M
            t7 = _mm_unpacklo_epi16(t8, rBuffer);
4573
2.31M
            x7 = _mm_unpackhi_epi16(t8, rBuffer);
4574
4575
4576
            /* add calculus by correct value : */
4577
4578
2.31M
            r1 = _mm_add_epi32(x1, x2);
4579
2.31M
            x3 = _mm_add_epi32(x3, x4);
4580
2.31M
            x5 = _mm_add_epi32(x5, x6);
4581
2.31M
            r1 = _mm_add_epi32(r1, x3);
4582
2.31M
            r1 = _mm_add_epi32(r1, x5);
4583
4584
2.31M
            r0 = _mm_add_epi32(t1, t2);
4585
2.31M
            t3 = _mm_add_epi32(t3, t4);
4586
2.31M
            t5 = _mm_add_epi32(t5, t6);
4587
2.31M
            r0 = _mm_add_epi32(r0, t3);
4588
2.31M
            r0 = _mm_add_epi32(r0, t5);
4589
2.31M
            r1 = _mm_add_epi32(r1, x7);
4590
2.31M
            r0 = _mm_add_epi32(r0, t7);
4591
2.31M
            r1 = _mm_srli_epi32(r1, 6);
4592
2.31M
            r0 = _mm_srli_epi32(r0, 6);
4593
4594
2.31M
            r1 = _mm_and_si128(r1,
4595
2.31M
                    _mm_set_epi16(0, -1, 0, -1, 0, -1, 0, -1));
4596
2.31M
            r0 = _mm_and_si128(r0,
4597
2.31M
                    _mm_set_epi16(0, -1, 0, -1, 0, -1, 0, -1));
4598
2.31M
            r0 = _mm_hadd_epi16(r0, r1);
4599
2.31M
            _mm_store_si128((__m128i *) &dst[x], r0);
4600
4601
2.31M
        }
4602
1.44M
        tmp += MAX_PB_SIZE;
4603
1.44M
        dst += dststride;
4604
1.44M
    }
4605
149k
}
4606
void ff_hevc_put_hevc_qpel_h_3_v_2_sse(int16_t *dst, ptrdiff_t dststride,
4607
                                       const uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
4608
144k
        int16_t* mcbuffer) {
4609
144k
    int x, y;
4610
144k
    uint8_t *src = (uint8_t*) _src;
4611
144k
    ptrdiff_t srcstride = _srcstride / sizeof(uint8_t);
4612
144k
    int16_t *tmp = mcbuffer;
4613
144k
    __m128i x1, x2, x3, x4, x5, x6, x7, x8, rBuffer, rTemp, r0, r1;
4614
144k
    __m128i t1, t2, t3, t4, t5, t6, t7, t8;
4615
4616
144k
    src -= qpel_extra_before[2] * srcstride;
4617
144k
    r0 = _mm_set_epi8(-1, 4, -10, 58, 17, -5, 1, 0, -1, 4, -10, 58, 17, -5, 1,
4618
144k
            0);
4619
4620
    /* LOAD src from memory to registers to limit memory bandwidth */
4621
144k
    if (width == 4) {
4622
4623
259k
        for (y = 0; y < height + qpel_extra[2]; y += 2) {
4624
            /* load data in register     */
4625
231k
            x1 = _mm_loadu_si128((__m128i *) &src[-2]);
4626
231k
            x1 = _mm_slli_si128(x1, 1);
4627
231k
            src += srcstride;
4628
231k
            t1 = _mm_loadu_si128((__m128i *) &src[-2]);
4629
231k
            t1 = _mm_slli_si128(t1, 1);
4630
231k
            x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
4631
231k
            t2 = _mm_unpacklo_epi64(t1, _mm_srli_si128(t1, 1));
4632
231k
            x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
4633
231k
                    _mm_srli_si128(x1, 3));
4634
231k
            t3 = _mm_unpacklo_epi64(_mm_srli_si128(t1, 2),
4635
231k
                    _mm_srli_si128(t1, 3));
4636
4637
            /*  PMADDUBSW then PMADDW     */
4638
231k
            x2 = _mm_maddubs_epi16(x2, r0);
4639
231k
            t2 = _mm_maddubs_epi16(t2, r0);
4640
231k
            x3 = _mm_maddubs_epi16(x3, r0);
4641
231k
            t3 = _mm_maddubs_epi16(t3, r0);
4642
231k
            x2 = _mm_hadd_epi16(x2, x3);
4643
231k
            t2 = _mm_hadd_epi16(t2, t3);
4644
231k
            x2 = _mm_hadd_epi16(x2, _mm_set1_epi16(0));
4645
231k
            t2 = _mm_hadd_epi16(t2, _mm_set1_epi16(0));
4646
231k
            x2 = _mm_srli_epi16(x2, BIT_DEPTH - 8);
4647
231k
            t2 = _mm_srli_epi16(t2, BIT_DEPTH - 8);
4648
            /* give results back            */
4649
231k
            _mm_storel_epi64((__m128i *) &tmp[0], x2);
4650
4651
231k
            tmp += MAX_PB_SIZE;
4652
231k
            _mm_storel_epi64((__m128i *) &tmp[0], t2);
4653
4654
231k
            src += srcstride;
4655
231k
            tmp += MAX_PB_SIZE;
4656
231k
        }
4657
28.6k
    } else
4658
2.09M
        for (y = 0; y < height + qpel_extra[2]; y++) {
4659
5.03M
            for (x = 0; x < width; x += 8) {
4660
                /* load data in register     */
4661
3.06M
                x1 = _mm_loadu_si128((__m128i *) &src[x - 2]);
4662
3.06M
                x1 = _mm_slli_si128(x1, 1);
4663
3.06M
                x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
4664
3.06M
                x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
4665
3.06M
                        _mm_srli_si128(x1, 3));
4666
3.06M
                x4 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 4),
4667
3.06M
                        _mm_srli_si128(x1, 5));
4668
3.06M
                x5 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 6),
4669
3.06M
                        _mm_srli_si128(x1, 7));
4670
4671
                /*  PMADDUBSW then PMADDW     */
4672
3.06M
                x2 = _mm_maddubs_epi16(x2, r0);
4673
3.06M
                x3 = _mm_maddubs_epi16(x3, r0);
4674
3.06M
                x4 = _mm_maddubs_epi16(x4, r0);
4675
3.06M
                x5 = _mm_maddubs_epi16(x5, r0);
4676
3.06M
                x2 = _mm_hadd_epi16(x2, x3);
4677
3.06M
                x4 = _mm_hadd_epi16(x4, x5);
4678
3.06M
                x2 = _mm_hadd_epi16(x2, x4);
4679
3.06M
                x2 = _mm_srli_si128(x2, BIT_DEPTH - 8);
4680
4681
                /* give results back            */
4682
3.06M
                _mm_store_si128((__m128i *) &tmp[x], x2);
4683
4684
3.06M
            }
4685
1.97M
            src += srcstride;
4686
1.97M
            tmp += MAX_PB_SIZE;
4687
1.97M
        }
4688
4689
144k
    tmp = mcbuffer + qpel_extra_before[2] * MAX_PB_SIZE;
4690
144k
    srcstride = MAX_PB_SIZE;
4691
4692
    /* vertical treatment on temp table : tmp contains 16 bit values, so need to use 32 bit  integers
4693
     for register calculations */
4694
144k
    rTemp = _mm_set_epi16(-1, 4, -11, 40, 40, -11, 4, -1);
4695
1.53M
    for (y = 0; y < height; y++) {
4696
3.60M
        for (x = 0; x < width; x += 8) {
4697
4698
2.21M
            x1 = _mm_load_si128((__m128i *) &tmp[x - 3 * srcstride]);
4699
2.21M
            x2 = _mm_load_si128((__m128i *) &tmp[x - 2 * srcstride]);
4700
2.21M
            x3 = _mm_load_si128((__m128i *) &tmp[x - srcstride]);
4701
2.21M
            x4 = _mm_load_si128((__m128i *) &tmp[x]);
4702
2.21M
            x5 = _mm_load_si128((__m128i *) &tmp[x + srcstride]);
4703
2.21M
            x6 = _mm_load_si128((__m128i *) &tmp[x + 2 * srcstride]);
4704
2.21M
            x7 = _mm_load_si128((__m128i *) &tmp[x + 3 * srcstride]);
4705
2.21M
            x8 = _mm_load_si128((__m128i *) &tmp[x + 4 * srcstride]);
4706
4707
2.21M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 0));
4708
2.21M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 1));
4709
2.21M
            t8 = _mm_mullo_epi16(x1, r0);
4710
2.21M
            rBuffer = _mm_mulhi_epi16(x1, r0);
4711
2.21M
            t7 = _mm_mullo_epi16(x2, r1);
4712
2.21M
            t1 = _mm_unpacklo_epi16(t8, rBuffer);
4713
2.21M
            x1 = _mm_unpackhi_epi16(t8, rBuffer);
4714
4715
2.21M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 2));
4716
2.21M
            rBuffer = _mm_mulhi_epi16(x2, r1);
4717
2.21M
            t8 = _mm_mullo_epi16(x3, r0);
4718
2.21M
            t2 = _mm_unpacklo_epi16(t7, rBuffer);
4719
2.21M
            x2 = _mm_unpackhi_epi16(t7, rBuffer);
4720
4721
2.21M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 3));
4722
2.21M
            rBuffer = _mm_mulhi_epi16(x3, r0);
4723
2.21M
            t7 = _mm_mullo_epi16(x4, r1);
4724
2.21M
            t3 = _mm_unpacklo_epi16(t8, rBuffer);
4725
2.21M
            x3 = _mm_unpackhi_epi16(t8, rBuffer);
4726
4727
2.21M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 4));
4728
2.21M
            rBuffer = _mm_mulhi_epi16(x4, r1);
4729
2.21M
            t8 = _mm_mullo_epi16(x5, r0);
4730
2.21M
            t4 = _mm_unpacklo_epi16(t7, rBuffer);
4731
2.21M
            x4 = _mm_unpackhi_epi16(t7, rBuffer);
4732
4733
2.21M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 5));
4734
2.21M
            rBuffer = _mm_mulhi_epi16(x5, r0);
4735
2.21M
            t7 = _mm_mullo_epi16(x6, r1);
4736
2.21M
            t5 = _mm_unpacklo_epi16(t8, rBuffer);
4737
2.21M
            x5 = _mm_unpackhi_epi16(t8, rBuffer);
4738
4739
2.21M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 6));
4740
2.21M
            rBuffer = _mm_mulhi_epi16(x6, r1);
4741
2.21M
            t8 = _mm_mullo_epi16(x7, r0);
4742
2.21M
            t6 = _mm_unpacklo_epi16(t7, rBuffer);
4743
2.21M
            x6 = _mm_unpackhi_epi16(t7, rBuffer);
4744
4745
2.21M
            rBuffer = _mm_mulhi_epi16(x7, r0);
4746
2.21M
            t7 = _mm_unpacklo_epi16(t8, rBuffer);
4747
2.21M
            x7 = _mm_unpackhi_epi16(t8, rBuffer);
4748
4749
2.21M
            t8 = _mm_unpacklo_epi16(
4750
2.21M
                    _mm_mullo_epi16(x8,
4751
2.21M
                            _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))),
4752
2.21M
                            _mm_mulhi_epi16(x8,
4753
2.21M
                                    _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))));
4754
2.21M
            x8 = _mm_unpackhi_epi16(
4755
2.21M
                    _mm_mullo_epi16(x8,
4756
2.21M
                            _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))),
4757
2.21M
                            _mm_mulhi_epi16(x8,
4758
2.21M
                                    _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))));
4759
4760
            /* add calculus by correct value : */
4761
4762
2.21M
            r1 = _mm_add_epi32(x1, x2);
4763
2.21M
            x3 = _mm_add_epi32(x3, x4);
4764
2.21M
            x5 = _mm_add_epi32(x5, x6);
4765
2.21M
            r1 = _mm_add_epi32(r1, x3);
4766
2.21M
            x7 = _mm_add_epi32(x7, x8);
4767
2.21M
            r1 = _mm_add_epi32(r1, x5);
4768
4769
2.21M
            r0 = _mm_add_epi32(t1, t2);
4770
2.21M
            t3 = _mm_add_epi32(t3, t4);
4771
2.21M
            t5 = _mm_add_epi32(t5, t6);
4772
2.21M
            r0 = _mm_add_epi32(r0, t3);
4773
2.21M
            t7 = _mm_add_epi32(t7, t8);
4774
2.21M
            r0 = _mm_add_epi32(r0, t5);
4775
2.21M
            r1 = _mm_add_epi32(r1, x7);
4776
2.21M
            r0 = _mm_add_epi32(r0, t7);
4777
2.21M
            r1 = _mm_srli_epi32(r1, 6);
4778
2.21M
            r0 = _mm_srli_epi32(r0, 6);
4779
4780
2.21M
            r1 = _mm_and_si128(r1,
4781
2.21M
                    _mm_set_epi16(0, -1, 0, -1, 0, -1, 0, -1));
4782
2.21M
            r0 = _mm_and_si128(r0,
4783
2.21M
                    _mm_set_epi16(0, -1, 0, -1, 0, -1, 0, -1));
4784
2.21M
            r0 = _mm_hadd_epi16(r0, r1);
4785
2.21M
            _mm_store_si128((__m128i *) &dst[x], r0);
4786
4787
2.21M
        }
4788
1.39M
        tmp += MAX_PB_SIZE;
4789
1.39M
        dst += dststride;
4790
1.39M
    }
4791
144k
}
4792
void ff_hevc_put_hevc_qpel_h_3_v_3_sse(int16_t *dst, ptrdiff_t dststride,
4793
                                       const uint8_t *_src, ptrdiff_t _srcstride, int width, int height,
4794
235k
        int16_t* mcbuffer) {
4795
235k
    int x, y;
4796
235k
    uint8_t *src = (uint8_t*) _src;
4797
235k
    ptrdiff_t srcstride = _srcstride / sizeof(uint8_t);
4798
235k
    int16_t *tmp = mcbuffer;
4799
235k
    __m128i x1, x2, x3, x4, x5, x6, x7, x8, rBuffer, rTemp, r0, r1;
4800
235k
    __m128i t1, t2, t3, t4, t5, t6, t7, t8;
4801
4802
235k
    src -= qpel_extra_before[3] * srcstride;
4803
235k
    r0 = _mm_set_epi8(-1, 4, -10, 58, 17, -5, 1, 0, -1, 4, -10, 58, 17, -5, 1,
4804
235k
            0);
4805
4806
    /* LOAD src from memory to registers to limit memory bandwidth */
4807
235k
    if (width == 4) {
4808
4809
393k
        for (y = 0; y < height + qpel_extra[3]; y += 2) {
4810
            /* load data in register     */
4811
344k
            x1 = _mm_loadu_si128((__m128i *) &src[-2]);
4812
344k
            x1 = _mm_slli_si128(x1, 1);
4813
344k
            src += srcstride;
4814
344k
            t1 = _mm_loadu_si128((__m128i *) &src[-2]);
4815
344k
            t1 = _mm_slli_si128(t1, 1);
4816
344k
            x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
4817
344k
            t2 = _mm_unpacklo_epi64(t1, _mm_srli_si128(t1, 1));
4818
344k
            x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
4819
344k
                    _mm_srli_si128(x1, 3));
4820
344k
            t3 = _mm_unpacklo_epi64(_mm_srli_si128(t1, 2),
4821
344k
                    _mm_srli_si128(t1, 3));
4822
4823
            /*  PMADDUBSW then PMADDW     */
4824
344k
            x2 = _mm_maddubs_epi16(x2, r0);
4825
344k
            t2 = _mm_maddubs_epi16(t2, r0);
4826
344k
            x3 = _mm_maddubs_epi16(x3, r0);
4827
344k
            t3 = _mm_maddubs_epi16(t3, r0);
4828
344k
            x2 = _mm_hadd_epi16(x2, x3);
4829
344k
            t2 = _mm_hadd_epi16(t2, t3);
4830
344k
            x2 = _mm_hadd_epi16(x2, _mm_set1_epi16(0));
4831
344k
            t2 = _mm_hadd_epi16(t2, _mm_set1_epi16(0));
4832
344k
            x2 = _mm_srli_epi16(x2, BIT_DEPTH - 8);
4833
344k
            t2 = _mm_srli_epi16(t2, BIT_DEPTH - 8);
4834
            /* give results back            */
4835
344k
            _mm_storel_epi64((__m128i *) &tmp[0], x2);
4836
4837
344k
            tmp += MAX_PB_SIZE;
4838
344k
            _mm_storel_epi64((__m128i *) &tmp[0], t2);
4839
4840
344k
            src += srcstride;
4841
344k
            tmp += MAX_PB_SIZE;
4842
344k
        }
4843
48.2k
    } else
4844
3.11M
        for (y = 0; y < height + qpel_extra[3]; y++) {
4845
7.50M
            for (x = 0; x < width; x += 8) {
4846
                /* load data in register     */
4847
4.57M
                x1 = _mm_loadu_si128((__m128i *) &src[x - 2]);
4848
4.57M
                x1 = _mm_slli_si128(x1, 1);
4849
4.57M
                x2 = _mm_unpacklo_epi64(x1, _mm_srli_si128(x1, 1));
4850
4.57M
                x3 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 2),
4851
4.57M
                        _mm_srli_si128(x1, 3));
4852
4.57M
                x4 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 4),
4853
4.57M
                        _mm_srli_si128(x1, 5));
4854
4.57M
                x5 = _mm_unpacklo_epi64(_mm_srli_si128(x1, 6),
4855
4.57M
                        _mm_srli_si128(x1, 7));
4856
4857
                /*  PMADDUBSW then PMADDW     */
4858
4.57M
                x2 = _mm_maddubs_epi16(x2, r0);
4859
4.57M
                x3 = _mm_maddubs_epi16(x3, r0);
4860
4.57M
                x4 = _mm_maddubs_epi16(x4, r0);
4861
4.57M
                x5 = _mm_maddubs_epi16(x5, r0);
4862
4.57M
                x2 = _mm_hadd_epi16(x2, x3);
4863
4.57M
                x4 = _mm_hadd_epi16(x4, x5);
4864
4.57M
                x2 = _mm_hadd_epi16(x2, x4);
4865
4.57M
                x2 = _mm_srli_si128(x2, BIT_DEPTH - 8);
4866
4867
                /* give results back            */
4868
4.57M
                _mm_store_si128((__m128i *) &tmp[x], x2);
4869
4870
4.57M
            }
4871
2.92M
            src += srcstride;
4872
2.92M
            tmp += MAX_PB_SIZE;
4873
2.92M
        }
4874
4875
235k
    tmp = mcbuffer + qpel_extra_before[3] * MAX_PB_SIZE;
4876
235k
    srcstride = MAX_PB_SIZE;
4877
4878
    /* vertical treatment on temp table : tmp contains 16 bit values, so need to use 32 bit  integers
4879
     for register calculations */
4880
235k
    rTemp = _mm_set_epi16(-1, 4, -10, 58, 17, -5, 1, 0);
4881
2.43M
    for (y = 0; y < height; y++) {
4882
5.70M
        for (x = 0; x < width; x += 8) {
4883
4884
3.49M
            x1 = _mm_setzero_si128();
4885
3.49M
            x2 = _mm_load_si128((__m128i *) &tmp[x - 2 * srcstride]);
4886
3.49M
            x3 = _mm_load_si128((__m128i *) &tmp[x - srcstride]);
4887
3.49M
            x4 = _mm_load_si128((__m128i *) &tmp[x]);
4888
3.49M
            x5 = _mm_load_si128((__m128i *) &tmp[x + srcstride]);
4889
3.49M
            x6 = _mm_load_si128((__m128i *) &tmp[x + 2 * srcstride]);
4890
3.49M
            x7 = _mm_load_si128((__m128i *) &tmp[x + 3 * srcstride]);
4891
3.49M
            x8 = _mm_load_si128((__m128i *) &tmp[x + 4 * srcstride]);
4892
4893
3.49M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 1));
4894
3.49M
            t7 = _mm_mullo_epi16(x2, r1);
4895
4896
4897
3.49M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 2));
4898
3.49M
            rBuffer = _mm_mulhi_epi16(x2, r1);
4899
3.49M
            t8 = _mm_mullo_epi16(x3, r0);
4900
3.49M
            t2 = _mm_unpacklo_epi16(t7, rBuffer);
4901
3.49M
            x2 = _mm_unpackhi_epi16(t7, rBuffer);
4902
4903
3.49M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 3));
4904
3.49M
            rBuffer = _mm_mulhi_epi16(x3, r0);
4905
3.49M
            t7 = _mm_mullo_epi16(x4, r1);
4906
3.49M
            t3 = _mm_unpacklo_epi16(t8, rBuffer);
4907
3.49M
            x3 = _mm_unpackhi_epi16(t8, rBuffer);
4908
4909
3.49M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 4));
4910
3.49M
            rBuffer = _mm_mulhi_epi16(x4, r1);
4911
3.49M
            t8 = _mm_mullo_epi16(x5, r0);
4912
3.49M
            t4 = _mm_unpacklo_epi16(t7, rBuffer);
4913
3.49M
            x4 = _mm_unpackhi_epi16(t7, rBuffer);
4914
4915
3.49M
            r1 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 5));
4916
3.49M
            rBuffer = _mm_mulhi_epi16(x5, r0);
4917
3.49M
            t7 = _mm_mullo_epi16(x6, r1);
4918
3.49M
            t5 = _mm_unpacklo_epi16(t8, rBuffer);
4919
3.49M
            x5 = _mm_unpackhi_epi16(t8, rBuffer);
4920
4921
3.49M
            r0 = _mm_set1_epi16(_mm_extract_epi16(rTemp, 6));
4922
3.49M
            rBuffer = _mm_mulhi_epi16(x6, r1);
4923
3.49M
            t8 = _mm_mullo_epi16(x7, r0);
4924
3.49M
            t6 = _mm_unpacklo_epi16(t7, rBuffer);
4925
3.49M
            x6 = _mm_unpackhi_epi16(t7, rBuffer);
4926
4927
3.49M
            rBuffer = _mm_mulhi_epi16(x7, r0);
4928
3.49M
            t7 = _mm_unpacklo_epi16(t8, rBuffer);
4929
3.49M
            x7 = _mm_unpackhi_epi16(t8, rBuffer);
4930
4931
3.49M
            t8 = _mm_unpacklo_epi16(
4932
3.49M
                    _mm_mullo_epi16(x8,
4933
3.49M
                            _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))),
4934
3.49M
                            _mm_mulhi_epi16(x8,
4935
3.49M
                                    _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))));
4936
3.49M
            x8 = _mm_unpackhi_epi16(
4937
3.49M
                    _mm_mullo_epi16(x8,
4938
3.49M
                            _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))),
4939
3.49M
                            _mm_mulhi_epi16(x8,
4940
3.49M
                                    _mm_set1_epi16(_mm_extract_epi16(rTemp, 7))));
4941
4942
            /* add calculus by correct value : */
4943
4944
3.49M
            x3 = _mm_add_epi32(x3, x4);
4945
3.49M
            x5 = _mm_add_epi32(x5, x6);
4946
3.49M
            r1 = _mm_add_epi32(x2, x3);
4947
3.49M
            x7 = _mm_add_epi32(x7, x8);
4948
3.49M
            r1 = _mm_add_epi32(r1, x5);
4949
4950
3.49M
            t3 = _mm_add_epi32(t3, t4);
4951
3.49M
            t5 = _mm_add_epi32(t5, t6);
4952
3.49M
            r0 = _mm_add_epi32(t2, t3);
4953
3.49M
            t7 = _mm_add_epi32(t7, t8);
4954
3.49M
            r0 = _mm_add_epi32(r0, t5);
4955
3.49M
            r1 = _mm_add_epi32(r1, x7);
4956
3.49M
            r0 = _mm_add_epi32(r0, t7);
4957
3.49M
            r1 = _mm_srli_epi32(r1, 6);
4958
3.49M
            r0 = _mm_srli_epi32(r0, 6);
4959
4960
3.49M
            r1 = _mm_and_si128(r1,
4961
3.49M
                    _mm_set_epi16(0, -1, 0, -1, 0, -1, 0, -1));
4962
3.49M
            r0 = _mm_and_si128(r0,
4963
3.49M
                    _mm_set_epi16(0, -1, 0, -1, 0, -1, 0, -1));
4964
3.49M
            r0 = _mm_hadd_epi16(r0, r1);
4965
3.49M
            _mm_store_si128((__m128i *) &dst[x], r0);
4966
4967
3.49M
        }
4968
2.20M
        tmp += MAX_PB_SIZE;
4969
2.20M
        dst += dststride;
4970
2.20M
    }
4971
235k
}