Coverage Report

Created: 2026-09-07 06:44

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/aom/av1/encoder/x86/reconinter_enc_sse2.c
Line
Count
Source
1
/*
2
 * Copyright (c) 2021, Alliance for Open Media. All rights reserved.
3
 *
4
 * This source code is subject to the terms of the BSD 2 Clause License and
5
 * the Alliance for Open Media Patent License 1.0. If the BSD 2 Clause License
6
 * was not distributed with this source code in the LICENSE file, you can
7
 * obtain it at www.aomedia.org/license/software. If the Alliance for Open
8
 * Media Patent License 1.0 was not distributed with this source code in the
9
 * PATENTS file, you can obtain it at www.aomedia.org/license/patent.
10
 */
11
12
#include <assert.h>
13
#include <emmintrin.h>  // SSE2
14
15
#include "config/aom_config.h"
16
#include "config/aom_dsp_rtcd.h"
17
#include "config/aom_scale_rtcd.h"
18
19
#include "aom/aom_integer.h"
20
#include "aom_dsp/blend.h"
21
#include "aom_dsp/x86/mem_sse2.h"
22
#include "aom_dsp/x86/synonyms.h"
23
24
#include "av1/common/av1_common_int.h"
25
#include "av1/common/blockd.h"
26
#include "av1/common/mvref_common.h"
27
#include "av1/common/obmc.h"
28
#include "av1/common/reconinter.h"
29
#include "av1/common/reconintra.h"
30
#include "av1/encoder/reconinter_enc.h"
31
32
void aom_upsampled_pred_sse2(MACROBLOCKD *xd, const struct AV1Common *const cm,
33
                             int mi_row, int mi_col, const MV *const mv,
34
                             uint8_t *comp_pred, int width, int height,
35
                             int subpel_x_q3, int subpel_y_q3,
36
                             const uint8_t *ref, int ref_stride,
37
0
                             int subpel_search) {
38
0
  if (aom_upsampled_pred_scaled(xd, cm, mi_row, mi_col, mv, comp_pred, width,
39
0
                                height)) {
40
0
    return;
41
0
  }
42
43
0
  const InterpFilterParams *filter = av1_get_filter(subpel_search);
44
  // (TODO:yunqing) 2-tap case uses 4-tap functions since there is no SIMD for
45
  // 2-tap yet.
46
0
  int filter_taps = (subpel_search <= USE_4_TAPS) ? 4 : SUBPEL_TAPS;
47
48
0
  if (!subpel_x_q3 && !subpel_y_q3) {
49
0
    if (width >= 16) {
50
0
      int i;
51
0
      assert(!(width & 15));
52
      /*Read 16 pixels one row at a time.*/
53
0
      for (i = 0; i < height; i++) {
54
0
        int j;
55
0
        for (j = 0; j < width; j += 16) {
56
0
          xx_storeu_128(comp_pred, xx_loadu_128(ref));
57
0
          comp_pred += 16;
58
0
          ref += 16;
59
0
        }
60
0
        ref += ref_stride - width;
61
0
      }
62
0
    } else if (width >= 8) {
63
0
      int i;
64
0
      assert(!(width & 7));
65
0
      assert(!(height & 1));
66
      /*Read 8 pixels two rows at a time.*/
67
0
      for (i = 0; i < height; i += 2) {
68
0
        __m128i s0 = xx_loadl_64(ref + 0 * ref_stride);
69
0
        __m128i s1 = xx_loadl_64(ref + 1 * ref_stride);
70
0
        xx_storeu_128(comp_pred, _mm_unpacklo_epi64(s0, s1));
71
0
        comp_pred += 16;
72
0
        ref += 2 * ref_stride;
73
0
      }
74
0
    } else {
75
0
      int i;
76
0
      assert(!(width & 3));
77
0
      assert(!(height & 3));
78
      /*Read 4 pixels four rows at a time.*/
79
0
      for (i = 0; i < height; i++) {
80
0
        const __m128i row0 = xx_loadl_64(ref + 0 * ref_stride);
81
0
        const __m128i row1 = xx_loadl_64(ref + 1 * ref_stride);
82
0
        const __m128i row2 = xx_loadl_64(ref + 2 * ref_stride);
83
0
        const __m128i row3 = xx_loadl_64(ref + 3 * ref_stride);
84
0
        const __m128i reg = _mm_unpacklo_epi64(_mm_unpacklo_epi32(row0, row1),
85
0
                                               _mm_unpacklo_epi32(row2, row3));
86
0
        xx_storeu_128(comp_pred, reg);
87
0
        comp_pred += 16;
88
0
        ref += 4 * ref_stride;
89
0
      }
90
0
    }
91
0
  } else if (!subpel_y_q3) {
92
0
    const int16_t *const kernel =
93
0
        av1_get_interp_filter_subpel_kernel(filter, subpel_x_q3 << 1);
94
0
    aom_convolve8_horiz(ref, ref_stride, comp_pred, width, kernel, 16, NULL, -1,
95
0
                        width, height);
96
0
  } else if (!subpel_x_q3) {
97
0
    const int16_t *const kernel =
98
0
        av1_get_interp_filter_subpel_kernel(filter, subpel_y_q3 << 1);
99
0
    aom_convolve8_vert(ref, ref_stride, comp_pred, width, NULL, -1, kernel, 16,
100
0
                       width, height);
101
0
  } else {
102
0
    uint8_t *temp = comp_pred;
103
0
    const int16_t *const kernel_x =
104
0
        av1_get_interp_filter_subpel_kernel(filter, subpel_x_q3 << 1);
105
0
    const int16_t *const kernel_y =
106
0
        av1_get_interp_filter_subpel_kernel(filter, subpel_y_q3 << 1);
107
0
    const uint8_t *ref_start = ref - ref_stride * ((filter_taps >> 1) - 1);
108
0
    uint8_t *temp_start_horiz = (subpel_search <= USE_4_TAPS)
109
0
                                    ? temp + (filter_taps >> 1) * MAX_SB_SIZE
110
0
                                    : temp;
111
0
    uint8_t *temp_start_vert = temp + MAX_SB_SIZE * ((filter->taps >> 1) - 1);
112
0
    int intermediate_height =
113
0
        (((height - 1) * 8 + subpel_y_q3) >> 3) + filter_taps;
114
0
    assert(intermediate_height < (MAX_SB_SIZE + SUBPEL_TAPS));
115
0
    aom_convolve8_horiz(ref_start, ref_stride, temp_start_horiz, MAX_SB_SIZE,
116
0
                        kernel_x, 16, NULL, -1, width, intermediate_height);
117
0
    aom_convolve8_vert(temp_start_vert, MAX_SB_SIZE, comp_pred, width, NULL, -1,
118
0
                       kernel_y, 16, width, height);
119
0
  }
120
0
}
121
122
#if CONFIG_AV1_HIGHBITDEPTH
123
void aom_highbd_upsampled_pred_sse2(MACROBLOCKD *xd,
124
                                    const struct AV1Common *const cm,
125
                                    int mi_row, int mi_col, const MV *const mv,
126
                                    uint8_t *comp_pred8, int width, int height,
127
                                    int subpel_x_q3, int subpel_y_q3,
128
                                    const uint8_t *ref8, int ref_stride, int bd,
129
0
                                    int subpel_search) {
130
0
  if (aom_upsampled_pred_scaled(xd, cm, mi_row, mi_col, mv, comp_pred8, width,
131
0
                                height)) {
132
0
    return;
133
0
  }
134
135
0
  const InterpFilterParams *filter = av1_get_filter(subpel_search);
136
0
  int filter_taps = (subpel_search <= USE_4_TAPS) ? 4 : SUBPEL_TAPS;
137
0
  if (!subpel_x_q3 && !subpel_y_q3) {
138
0
    const uint16_t *ref = CONVERT_TO_SHORTPTR(ref8);
139
0
    uint16_t *comp_pred = CONVERT_TO_SHORTPTR(comp_pred8);
140
0
    if (width >= 8) {
141
0
      int i;
142
0
      assert(!(width & 7));
143
      /*Read 8 pixels one row at a time.*/
144
0
      for (i = 0; i < height; i++) {
145
0
        int j;
146
0
        for (j = 0; j < width; j += 8) {
147
0
          __m128i s0 = _mm_loadu_si128((const __m128i *)ref);
148
0
          _mm_storeu_si128((__m128i *)comp_pred, s0);
149
0
          comp_pred += 8;
150
0
          ref += 8;
151
0
        }
152
0
        ref += ref_stride - width;
153
0
      }
154
0
    } else {
155
0
      int i;
156
0
      assert(!(width & 3));
157
      /*Read 4 pixels two rows at a time.*/
158
0
      for (i = 0; i < height; i += 2) {
159
0
        __m128i s0 = _mm_loadl_epi64((const __m128i *)ref);
160
0
        __m128i s1 = _mm_loadl_epi64((const __m128i *)(ref + ref_stride));
161
0
        __m128i t0 = _mm_unpacklo_epi64(s0, s1);
162
0
        _mm_storeu_si128((__m128i *)comp_pred, t0);
163
0
        comp_pred += 8;
164
0
        ref += 2 * ref_stride;
165
0
      }
166
0
    }
167
0
  } else if (!subpel_y_q3) {
168
0
    const int16_t *const kernel =
169
0
        av1_get_interp_filter_subpel_kernel(filter, subpel_x_q3 << 1);
170
0
    aom_highbd_convolve8_horiz(ref8, ref_stride, comp_pred8, width, kernel, 16,
171
0
                               NULL, -1, width, height, bd);
172
0
  } else if (!subpel_x_q3) {
173
0
    const int16_t *const kernel =
174
0
        av1_get_interp_filter_subpel_kernel(filter, subpel_y_q3 << 1);
175
0
    aom_highbd_convolve8_vert(ref8, ref_stride, comp_pred8, width, NULL, -1,
176
0
                              kernel, 16, width, height, bd);
177
0
  } else {
178
    // The src and dst parameters of 'aom_highbd_convolve8_vert()' call are the
179
    // same buffer ('comp_pred8') although they start from different offsets in
180
    // the buffer. For any given row 'y', the 8-tap vertical filter reads a
181
    // window of input rows from 'y-3' to 'y+4' (stride = MAX_SB_SIZE) and the
182
    // result is written to row 'y-3' (stride = width) of the `comp_pred8`
183
    // buffer. Since 'width <= MAX_SB_SIZE', output is written only after its
184
    // corresponding input has already been read, eliminating any risk of
185
    // overwriting data required for future iterations. The function
186
    // 'aom_highbd_convolve8_vert()' must process the rows from top to bottom in
187
    // increading order of row index, otherwise it will break the in-place
188
    // processing of 'comp_pred8'.
189
0
    uint16_t *temp = CONVERT_TO_SHORTPTR(comp_pred8);
190
0
    const uint16_t *ref = CONVERT_TO_SHORTPTR(ref8);
191
0
    const int16_t *const kernel_x =
192
0
        av1_get_interp_filter_subpel_kernel(filter, subpel_x_q3 << 1);
193
0
    const int16_t *const kernel_y =
194
0
        av1_get_interp_filter_subpel_kernel(filter, subpel_y_q3 << 1);
195
0
    const uint16_t *ref_start = ref - ref_stride * ((filter_taps >> 1) - 1);
196
0
    uint16_t *temp_start_horiz = (subpel_search <= USE_4_TAPS)
197
0
                                     ? temp + (filter_taps >> 1) * MAX_SB_SIZE
198
0
                                     : temp;
199
0
    uint16_t *temp_start_vert = temp + MAX_SB_SIZE * ((filter->taps >> 1) - 1);
200
0
    const int intermediate_height =
201
0
        (((height - 1) * 8 + subpel_y_q3) >> 3) + filter_taps;
202
0
    assert(intermediate_height < (MAX_SB_SIZE + SUBPEL_TAPS));
203
0
    aom_highbd_convolve8_horiz(CONVERT_TO_BYTEPTR(ref_start), ref_stride,
204
0
                               CONVERT_TO_BYTEPTR(temp_start_horiz),
205
0
                               MAX_SB_SIZE, kernel_x, 16, NULL, -1, width,
206
0
                               intermediate_height, bd);
207
0
    aom_highbd_convolve8_vert(CONVERT_TO_BYTEPTR(temp_start_vert), MAX_SB_SIZE,
208
0
                              comp_pred8, width, NULL, -1, kernel_y, 16, width,
209
0
                              height, bd);
210
0
  }
211
0
}
212
213
void aom_highbd_comp_avg_upsampled_pred_sse2(
214
    MACROBLOCKD *xd, const struct AV1Common *const cm, int mi_row, int mi_col,
215
    const MV *const mv, uint8_t *comp_pred8, const uint8_t *pred8, int width,
216
    int height, int subpel_x_q3, int subpel_y_q3, const uint8_t *ref8,
217
0
    int ref_stride, int bd, int subpel_search) {
218
0
  aom_highbd_upsampled_pred(xd, cm, mi_row, mi_col, mv, comp_pred8, width,
219
0
                            height, subpel_x_q3, subpel_y_q3, ref8, ref_stride,
220
0
                            bd, subpel_search);
221
0
  uint16_t *pred = CONVERT_TO_SHORTPTR(pred8);
222
0
  uint16_t *comp_pred16 = CONVERT_TO_SHORTPTR(comp_pred8);
223
  /*The total number of pixels must be a multiple of 8 (e.g., 4x4).*/
224
0
  assert(!(width * height & 7));
225
0
  int n = width * height >> 3;
226
0
  for (int i = 0; i < n; i++) {
227
0
    __m128i s0 = _mm_loadu_si128((const __m128i *)comp_pred16);
228
0
    __m128i p0 = _mm_loadu_si128((const __m128i *)pred);
229
0
    _mm_storeu_si128((__m128i *)comp_pred16, _mm_avg_epu16(s0, p0));
230
0
    comp_pred16 += 8;
231
0
    pred += 8;
232
0
  }
233
0
}
234
#endif  // CONFIG_AV1_HIGHBITDEPTH
235
236
void aom_comp_avg_upsampled_pred_sse2(
237
    MACROBLOCKD *xd, const struct AV1Common *const cm, int mi_row, int mi_col,
238
    const MV *const mv, uint8_t *comp_pred, const uint8_t *pred, int width,
239
    int height, int subpel_x_q3, int subpel_y_q3, const uint8_t *ref,
240
0
    int ref_stride, int subpel_search) {
241
0
  int n;
242
0
  int i;
243
0
  aom_upsampled_pred(xd, cm, mi_row, mi_col, mv, comp_pred, width, height,
244
0
                     subpel_x_q3, subpel_y_q3, ref, ref_stride, subpel_search);
245
  /*The total number of pixels must be a multiple of 16 (e.g., 4x4).*/
246
0
  assert(!(width * height & 15));
247
0
  n = width * height >> 4;
248
0
  for (i = 0; i < n; i++) {
249
0
    __m128i s0 = xx_loadu_128(comp_pred);
250
0
    __m128i p0 = xx_loadu_128(pred);
251
0
    xx_storeu_128(comp_pred, _mm_avg_epu8(s0, p0));
252
0
    comp_pred += 16;
253
0
    pred += 16;
254
0
  }
255
0
}