Coverage Report

Created: 2026-09-03 06:27

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/libavif/ext/libyuv/source/scale_common.cc
Line
Count
Source
1
/*
2
 *  Copyright 2013 The LibYuv Project Authors. All rights reserved.
3
 *
4
 *  Use of this source code is governed by a BSD-style license
5
 *  that can be found in the LICENSE file in the root of the source
6
 *  tree. An additional intellectual property rights grant can be found
7
 *  in the file PATENTS. All contributing project authors may
8
 *  be found in the AUTHORS file in the root of the source tree.
9
 */
10
11
#include "libyuv/scale.h"
12
13
#include <assert.h>
14
#include <string.h>
15
16
#include "libyuv/cpu_id.h"
17
#include "libyuv/planar_functions.h"  // For CopyARGB
18
#include "libyuv/row.h"
19
#include "libyuv/scale_row.h"
20
21
#ifdef __cplusplus
22
namespace libyuv {
23
extern "C" {
24
#endif
25
26
#ifdef __cplusplus
27
#define STATIC_CAST(type, expr) static_cast<type>(expr)
28
#else
29
#define STATIC_CAST(type, expr) (type)(expr)
30
#endif
31
32
50.7k
static __inline int Abs(int v) {
33
50.7k
  return v >= 0 ? v : -v;
34
50.7k
}
35
36
// CPU agnostic row functions
37
void ScaleRowDown2_C(const uint8_t* src_ptr,
38
                     ptrdiff_t src_stride,
39
                     uint8_t* dst,
40
0
                     int dst_width) {
41
0
  int x;
42
0
  (void)src_stride;
43
0
  for (x = 0; x < dst_width - 1; x += 2) {
44
0
    dst[0] = src_ptr[1];
45
0
    dst[1] = src_ptr[3];
46
0
    dst += 2;
47
0
    src_ptr += 4;
48
0
  }
49
0
  if (dst_width & 1) {
50
0
    dst[0] = src_ptr[1];
51
0
  }
52
0
}
53
54
void ScaleRowDown2_16_C(const uint16_t* src_ptr,
55
                        ptrdiff_t src_stride,
56
                        uint16_t* dst,
57
0
                        int dst_width) {
58
0
  int x;
59
0
  (void)src_stride;
60
0
  for (x = 0; x < dst_width - 1; x += 2) {
61
0
    dst[0] = src_ptr[1];
62
0
    dst[1] = src_ptr[3];
63
0
    dst += 2;
64
0
    src_ptr += 4;
65
0
  }
66
0
  if (dst_width & 1) {
67
0
    dst[0] = src_ptr[1];
68
0
  }
69
0
}
70
71
72
void ScaleRowDown2Linear_C(const uint8_t* src_ptr,
73
                           ptrdiff_t src_stride,
74
                           uint8_t* dst,
75
0
                           int dst_width) {
76
0
  const uint8_t* s = src_ptr;
77
0
  int x;
78
0
  (void)src_stride;
79
0
  for (x = 0; x < dst_width - 1; x += 2) {
80
0
    dst[0] = (s[0] + s[1] + 1) >> 1;
81
0
    dst[1] = (s[2] + s[3] + 1) >> 1;
82
0
    dst += 2;
83
0
    s += 4;
84
0
  }
85
0
  if (dst_width & 1) {
86
0
    dst[0] = (s[0] + s[1] + 1) >> 1;
87
0
  }
88
0
}
89
90
void ScaleRowDown2Linear_16_C(const uint16_t* src_ptr,
91
                              ptrdiff_t src_stride,
92
                              uint16_t* dst,
93
0
                              int dst_width) {
94
0
  const uint16_t* s = src_ptr;
95
0
  int x;
96
0
  (void)src_stride;
97
0
  for (x = 0; x < dst_width - 1; x += 2) {
98
0
    dst[0] = (s[0] + s[1] + 1) >> 1;
99
0
    dst[1] = (s[2] + s[3] + 1) >> 1;
100
0
    dst += 2;
101
0
    s += 4;
102
0
  }
103
0
  if (dst_width & 1) {
104
0
    dst[0] = (s[0] + s[1] + 1) >> 1;
105
0
  }
106
0
}
107
108
109
void ScaleRowDown2Box_C(const uint8_t* src_ptr,
110
                        ptrdiff_t src_stride,
111
                        uint8_t* dst,
112
807
                        int dst_width) {
113
807
  const uint8_t* s = src_ptr;
114
807
  const uint8_t* t = src_ptr + src_stride;
115
807
  int x;
116
7.11k
  for (x = 0; x < dst_width - 1; x += 2) {
117
6.30k
    dst[0] = (s[0] + s[1] + t[0] + t[1] + 2) >> 2;
118
6.30k
    dst[1] = (s[2] + s[3] + t[2] + t[3] + 2) >> 2;
119
6.30k
    dst += 2;
120
6.30k
    s += 4;
121
6.30k
    t += 4;
122
6.30k
  }
123
807
  if (dst_width & 1) {
124
160
    dst[0] = (s[0] + s[1] + t[0] + t[1] + 2) >> 2;
125
160
  }
126
807
}
127
128
void ScaleRowDown2Box_Odd_C(const uint8_t* src_ptr,
129
                            ptrdiff_t src_stride,
130
                            uint8_t* dst,
131
0
                            int dst_width) {
132
0
  const uint8_t* s = src_ptr;
133
0
  const uint8_t* t = src_ptr + src_stride;
134
0
  int x;
135
0
  dst_width -= 1;
136
0
  for (x = 0; x < dst_width - 1; x += 2) {
137
0
    dst[0] = (s[0] + s[1] + t[0] + t[1] + 2) >> 2;
138
0
    dst[1] = (s[2] + s[3] + t[2] + t[3] + 2) >> 2;
139
0
    dst += 2;
140
0
    s += 4;
141
0
    t += 4;
142
0
  }
143
0
  if (dst_width & 1) {
144
0
    dst[0] = (s[0] + s[1] + t[0] + t[1] + 2) >> 2;
145
0
    dst += 1;
146
0
    s += 2;
147
0
    t += 2;
148
0
  }
149
0
  dst[0] = (s[0] + t[0] + 1) >> 1;
150
0
}
151
152
void ScaleRowDown2Box_16_C(const uint16_t* src_ptr,
153
                           ptrdiff_t src_stride,
154
                           uint16_t* dst,
155
744
                           int dst_width) {
156
744
  const uint16_t* s = src_ptr;
157
744
  const uint16_t* t = src_ptr + src_stride;
158
744
  int x;
159
24.2k
  for (x = 0; x < dst_width - 1; x += 2) {
160
23.4k
    dst[0] = (s[0] + s[1] + t[0] + t[1] + 2) >> 2;
161
23.4k
    dst[1] = (s[2] + s[3] + t[2] + t[3] + 2) >> 2;
162
23.4k
    dst += 2;
163
23.4k
    s += 4;
164
23.4k
    t += 4;
165
23.4k
  }
166
744
  if (dst_width & 1) {
167
364
    dst[0] = (s[0] + s[1] + t[0] + t[1] + 2) >> 2;
168
364
  }
169
744
}
170
171
172
void ScaleRowDown4_C(const uint8_t* src_ptr,
173
                     ptrdiff_t src_stride,
174
                     uint8_t* dst,
175
0
                     int dst_width) {
176
0
  int x;
177
0
  (void)src_stride;
178
0
  for (x = 0; x < dst_width - 1; x += 2) {
179
0
    dst[0] = src_ptr[2];
180
0
    dst[1] = src_ptr[6];
181
0
    dst += 2;
182
0
    src_ptr += 8;
183
0
  }
184
0
  if (dst_width & 1) {
185
0
    dst[0] = src_ptr[2];
186
0
  }
187
0
}
188
189
void ScaleRowDown4_16_C(const uint16_t* src_ptr,
190
                        ptrdiff_t src_stride,
191
                        uint16_t* dst,
192
0
                        int dst_width) {
193
0
  int x;
194
0
  (void)src_stride;
195
0
  for (x = 0; x < dst_width - 1; x += 2) {
196
0
    dst[0] = src_ptr[2];
197
0
    dst[1] = src_ptr[6];
198
0
    dst += 2;
199
0
    src_ptr += 8;
200
0
  }
201
0
  if (dst_width & 1) {
202
0
    dst[0] = src_ptr[2];
203
0
  }
204
0
}
205
206
void ScaleRowDown4Box_C(const uint8_t* src_ptr,
207
                        ptrdiff_t src_stride,
208
                        uint8_t* dst,
209
495
                        int dst_width) {
210
495
  int x;
211
1.76k
  for (x = 0; x < dst_width - 1; x += 2) {
212
1.26k
    dst[0] = (src_ptr[0] + src_ptr[1] + src_ptr[2] + src_ptr[3] +
213
1.26k
              src_ptr[src_stride + 0] + src_ptr[src_stride + 1] +
214
1.26k
              src_ptr[src_stride + 2] + src_ptr[src_stride + 3] +
215
1.26k
              src_ptr[src_stride * 2 + 0] + src_ptr[src_stride * 2 + 1] +
216
1.26k
              src_ptr[src_stride * 2 + 2] + src_ptr[src_stride * 2 + 3] +
217
1.26k
              src_ptr[src_stride * 3 + 0] + src_ptr[src_stride * 3 + 1] +
218
1.26k
              src_ptr[src_stride * 3 + 2] + src_ptr[src_stride * 3 + 3] + 8) >>
219
1.26k
             4;
220
1.26k
    dst[1] = (src_ptr[4] + src_ptr[5] + src_ptr[6] + src_ptr[7] +
221
1.26k
              src_ptr[src_stride + 4] + src_ptr[src_stride + 5] +
222
1.26k
              src_ptr[src_stride + 6] + src_ptr[src_stride + 7] +
223
1.26k
              src_ptr[src_stride * 2 + 4] + src_ptr[src_stride * 2 + 5] +
224
1.26k
              src_ptr[src_stride * 2 + 6] + src_ptr[src_stride * 2 + 7] +
225
1.26k
              src_ptr[src_stride * 3 + 4] + src_ptr[src_stride * 3 + 5] +
226
1.26k
              src_ptr[src_stride * 3 + 6] + src_ptr[src_stride * 3 + 7] + 8) >>
227
1.26k
             4;
228
1.26k
    dst += 2;
229
1.26k
    src_ptr += 8;
230
1.26k
  }
231
495
  if (dst_width & 1) {
232
32
    dst[0] = (src_ptr[0] + src_ptr[1] + src_ptr[2] + src_ptr[3] +
233
32
              src_ptr[src_stride + 0] + src_ptr[src_stride + 1] +
234
32
              src_ptr[src_stride + 2] + src_ptr[src_stride + 3] +
235
32
              src_ptr[src_stride * 2 + 0] + src_ptr[src_stride * 2 + 1] +
236
32
              src_ptr[src_stride * 2 + 2] + src_ptr[src_stride * 2 + 3] +
237
32
              src_ptr[src_stride * 3 + 0] + src_ptr[src_stride * 3 + 1] +
238
32
              src_ptr[src_stride * 3 + 2] + src_ptr[src_stride * 3 + 3] + 8) >>
239
32
             4;
240
32
  }
241
495
}
242
243
void ScaleRowDown4Box_16_C(const uint16_t* src_ptr,
244
                           ptrdiff_t src_stride,
245
                           uint16_t* dst,
246
529
                           int dst_width) {
247
529
  int x;
248
8.01k
  for (x = 0; x < dst_width - 1; x += 2) {
249
7.48k
    dst[0] = (src_ptr[0] + src_ptr[1] + src_ptr[2] + src_ptr[3] +
250
7.48k
              src_ptr[src_stride + 0] + src_ptr[src_stride + 1] +
251
7.48k
              src_ptr[src_stride + 2] + src_ptr[src_stride + 3] +
252
7.48k
              src_ptr[src_stride * 2 + 0] + src_ptr[src_stride * 2 + 1] +
253
7.48k
              src_ptr[src_stride * 2 + 2] + src_ptr[src_stride * 2 + 3] +
254
7.48k
              src_ptr[src_stride * 3 + 0] + src_ptr[src_stride * 3 + 1] +
255
7.48k
              src_ptr[src_stride * 3 + 2] + src_ptr[src_stride * 3 + 3] + 8) >>
256
7.48k
             4;
257
7.48k
    dst[1] = (src_ptr[4] + src_ptr[5] + src_ptr[6] + src_ptr[7] +
258
7.48k
              src_ptr[src_stride + 4] + src_ptr[src_stride + 5] +
259
7.48k
              src_ptr[src_stride + 6] + src_ptr[src_stride + 7] +
260
7.48k
              src_ptr[src_stride * 2 + 4] + src_ptr[src_stride * 2 + 5] +
261
7.48k
              src_ptr[src_stride * 2 + 6] + src_ptr[src_stride * 2 + 7] +
262
7.48k
              src_ptr[src_stride * 3 + 4] + src_ptr[src_stride * 3 + 5] +
263
7.48k
              src_ptr[src_stride * 3 + 6] + src_ptr[src_stride * 3 + 7] + 8) >>
264
7.48k
             4;
265
7.48k
    dst += 2;
266
7.48k
    src_ptr += 8;
267
7.48k
  }
268
529
  if (dst_width & 1) {
269
30
    dst[0] = (src_ptr[0] + src_ptr[1] + src_ptr[2] + src_ptr[3] +
270
30
              src_ptr[src_stride + 0] + src_ptr[src_stride + 1] +
271
30
              src_ptr[src_stride + 2] + src_ptr[src_stride + 3] +
272
30
              src_ptr[src_stride * 2 + 0] + src_ptr[src_stride * 2 + 1] +
273
30
              src_ptr[src_stride * 2 + 2] + src_ptr[src_stride * 2 + 3] +
274
30
              src_ptr[src_stride * 3 + 0] + src_ptr[src_stride * 3 + 1] +
275
30
              src_ptr[src_stride * 3 + 2] + src_ptr[src_stride * 3 + 3] + 8) >>
276
30
             4;
277
30
  }
278
529
}
279
280
void ScaleRowDown34_C(const uint8_t* src_ptr,
281
                      ptrdiff_t src_stride,
282
                      uint8_t* dst,
283
0
                      int dst_width) {
284
0
  int x;
285
0
  (void)src_stride;
286
0
  assert((dst_width % 3 == 0) && (dst_width > 0));
287
0
  for (x = 0; x < dst_width; x += 3) {
288
0
    dst[0] = src_ptr[0];
289
0
    dst[1] = src_ptr[1];
290
0
    dst[2] = src_ptr[3];
291
0
    dst += 3;
292
0
    src_ptr += 4;
293
0
  }
294
0
}
295
296
void ScaleRowDown34_16_C(const uint16_t* src_ptr,
297
                         ptrdiff_t src_stride,
298
                         uint16_t* dst,
299
0
                         int dst_width) {
300
0
  int x;
301
0
  (void)src_stride;
302
0
  assert((dst_width % 3 == 0) && (dst_width > 0));
303
0
  for (x = 0; x < dst_width; x += 3) {
304
0
    dst[0] = src_ptr[0];
305
0
    dst[1] = src_ptr[1];
306
0
    dst[2] = src_ptr[3];
307
0
    dst += 3;
308
0
    src_ptr += 4;
309
0
  }
310
0
}
311
312
// Filter rows 0 and 1 together, 3 : 1
313
void ScaleRowDown34_0_Box_C(const uint8_t* src_ptr,
314
                            ptrdiff_t src_stride,
315
                            uint8_t* d,
316
4
                            int dst_width) {
317
4
  const uint8_t* s = src_ptr;
318
4
  const uint8_t* t = src_ptr + src_stride;
319
4
  int x;
320
4
  assert((dst_width % 3 == 0) && (dst_width > 0));
321
8
  for (x = 0; x < dst_width; x += 3) {
322
4
    uint8_t a0 = (s[0] * 3 + s[1] * 1 + 2) >> 2;
323
4
    uint8_t a1 = (s[1] * 1 + s[2] * 1 + 1) >> 1;
324
4
    uint8_t a2 = (s[2] * 1 + s[3] * 3 + 2) >> 2;
325
4
    uint8_t b0 = (t[0] * 3 + t[1] * 1 + 2) >> 2;
326
4
    uint8_t b1 = (t[1] * 1 + t[2] * 1 + 1) >> 1;
327
4
    uint8_t b2 = (t[2] * 1 + t[3] * 3 + 2) >> 2;
328
4
    d[0] = (a0 * 3 + b0 + 2) >> 2;
329
4
    d[1] = (a1 * 3 + b1 + 2) >> 2;
330
4
    d[2] = (a2 * 3 + b2 + 2) >> 2;
331
4
    d += 3;
332
4
    s += 4;
333
4
    t += 4;
334
4
  }
335
4
}
336
337
void ScaleRowDown34_0_Box_16_C(const uint16_t* src_ptr,
338
                               ptrdiff_t src_stride,
339
                               uint16_t* d,
340
48
                               int dst_width) {
341
48
  const uint16_t* s = src_ptr;
342
48
  const uint16_t* t = src_ptr + src_stride;
343
48
  int x;
344
48
  assert((dst_width % 3 == 0) && (dst_width > 0));
345
228
  for (x = 0; x < dst_width; x += 3) {
346
180
    uint16_t a0 = (s[0] * 3 + s[1] * 1 + 2) >> 2;
347
180
    uint16_t a1 = (s[1] * 1 + s[2] * 1 + 1) >> 1;
348
180
    uint16_t a2 = (s[2] * 1 + s[3] * 3 + 2) >> 2;
349
180
    uint16_t b0 = (t[0] * 3 + t[1] * 1 + 2) >> 2;
350
180
    uint16_t b1 = (t[1] * 1 + t[2] * 1 + 1) >> 1;
351
180
    uint16_t b2 = (t[2] * 1 + t[3] * 3 + 2) >> 2;
352
180
    d[0] = (a0 * 3 + b0 + 2) >> 2;
353
180
    d[1] = (a1 * 3 + b1 + 2) >> 2;
354
180
    d[2] = (a2 * 3 + b2 + 2) >> 2;
355
180
    d += 3;
356
180
    s += 4;
357
180
    t += 4;
358
180
  }
359
48
}
360
361
// Filter rows 1 and 2 together, 1 : 1
362
void ScaleRowDown34_1_Box_C(const uint8_t* src_ptr,
363
                            ptrdiff_t src_stride,
364
                            uint8_t* d,
365
2
                            int dst_width) {
366
2
  const uint8_t* s = src_ptr;
367
2
  const uint8_t* t = src_ptr + src_stride;
368
2
  int x;
369
2
  assert((dst_width % 3 == 0) && (dst_width > 0));
370
4
  for (x = 0; x < dst_width; x += 3) {
371
2
    uint8_t a0 = (s[0] * 3 + s[1] * 1 + 2) >> 2;
372
2
    uint8_t a1 = (s[1] * 1 + s[2] * 1 + 1) >> 1;
373
2
    uint8_t a2 = (s[2] * 1 + s[3] * 3 + 2) >> 2;
374
2
    uint8_t b0 = (t[0] * 3 + t[1] * 1 + 2) >> 2;
375
2
    uint8_t b1 = (t[1] * 1 + t[2] * 1 + 1) >> 1;
376
2
    uint8_t b2 = (t[2] * 1 + t[3] * 3 + 2) >> 2;
377
2
    d[0] = (a0 + b0 + 1) >> 1;
378
2
    d[1] = (a1 + b1 + 1) >> 1;
379
2
    d[2] = (a2 + b2 + 1) >> 1;
380
2
    d += 3;
381
2
    s += 4;
382
2
    t += 4;
383
2
  }
384
2
}
385
386
void ScaleRowDown34_1_Box_16_C(const uint16_t* src_ptr,
387
                               ptrdiff_t src_stride,
388
                               uint16_t* d,
389
24
                               int dst_width) {
390
24
  const uint16_t* s = src_ptr;
391
24
  const uint16_t* t = src_ptr + src_stride;
392
24
  int x;
393
24
  assert((dst_width % 3 == 0) && (dst_width > 0));
394
114
  for (x = 0; x < dst_width; x += 3) {
395
90
    uint16_t a0 = (s[0] * 3 + s[1] * 1 + 2) >> 2;
396
90
    uint16_t a1 = (s[1] * 1 + s[2] * 1 + 1) >> 1;
397
90
    uint16_t a2 = (s[2] * 1 + s[3] * 3 + 2) >> 2;
398
90
    uint16_t b0 = (t[0] * 3 + t[1] * 1 + 2) >> 2;
399
90
    uint16_t b1 = (t[1] * 1 + t[2] * 1 + 1) >> 1;
400
90
    uint16_t b2 = (t[2] * 1 + t[3] * 3 + 2) >> 2;
401
90
    d[0] = (a0 + b0 + 1) >> 1;
402
90
    d[1] = (a1 + b1 + 1) >> 1;
403
90
    d[2] = (a2 + b2 + 1) >> 1;
404
90
    d += 3;
405
90
    s += 4;
406
90
    t += 4;
407
90
  }
408
24
}
409
410
// Sample position: (O is src sample position, X is dst sample position)
411
//
412
//      v dst_ptr at here           v stop at here
413
//  X O X   X O X   X O X   X O X   X O X
414
//    ^ src_ptr at here
415
void ScaleRowUp2_Linear_C(const uint8_t* src_ptr,
416
                          uint8_t* dst_ptr,
417
326k
                          int dst_width) {
418
326k
  int src_width = dst_width >> 1;
419
326k
  int x;
420
326k
  assert((dst_width % 2 == 0) && (dst_width >= 0));
421
1.61M
  for (x = 0; x < src_width; ++x) {
422
1.29M
    dst_ptr[2 * x + 0] = (src_ptr[x + 0] * 3 + src_ptr[x + 1] * 1 + 2) >> 2;
423
1.29M
    dst_ptr[2 * x + 1] = (src_ptr[x + 0] * 1 + src_ptr[x + 1] * 3 + 2) >> 2;
424
1.29M
  }
425
326k
}
426
427
// Sample position: (O is src sample position, X is dst sample position)
428
//
429
//    src_ptr at here
430
//  X v X   X   X   X   X   X   X   X   X
431
//    O       O       O       O       O
432
//  X   X   X   X   X   X   X   X   X   X
433
//      ^ dst_ptr at here           ^ stop at here
434
//  X   X   X   X   X   X   X   X   X   X
435
//    O       O       O       O       O
436
//  X   X   X   X   X   X   X   X   X   X
437
void ScaleRowUp2_Bilinear_C(const uint8_t* src_ptr,
438
                            ptrdiff_t src_stride,
439
                            uint8_t* dst_ptr,
440
                            ptrdiff_t dst_stride,
441
11.4k
                            int dst_width) {
442
11.4k
  const uint8_t* s = src_ptr;
443
11.4k
  const uint8_t* t = src_ptr + src_stride;
444
11.4k
  uint8_t* d = dst_ptr;
445
11.4k
  uint8_t* e = dst_ptr + dst_stride;
446
11.4k
  int src_width = dst_width >> 1;
447
11.4k
  int x;
448
11.4k
  assert((dst_width % 2 == 0) && (dst_width >= 0));
449
138k
  for (x = 0; x < src_width; ++x) {
450
127k
    d[2 * x + 0] =
451
127k
        (s[x + 0] * 9 + s[x + 1] * 3 + t[x + 0] * 3 + t[x + 1] * 1 + 8) >> 4;
452
127k
    d[2 * x + 1] =
453
127k
        (s[x + 0] * 3 + s[x + 1] * 9 + t[x + 0] * 1 + t[x + 1] * 3 + 8) >> 4;
454
127k
    e[2 * x + 0] =
455
127k
        (s[x + 0] * 3 + s[x + 1] * 1 + t[x + 0] * 9 + t[x + 1] * 3 + 8) >> 4;
456
127k
    e[2 * x + 1] =
457
127k
        (s[x + 0] * 1 + s[x + 1] * 3 + t[x + 0] * 3 + t[x + 1] * 9 + 8) >> 4;
458
127k
  }
459
11.4k
}
460
461
// Only suitable for at most 14 bit range.
462
void ScaleRowUp2_Linear_16_C(const uint16_t* src_ptr,
463
                             uint16_t* dst_ptr,
464
627k
                             int dst_width) {
465
627k
  int src_width = dst_width >> 1;
466
627k
  int x;
467
627k
  assert((dst_width % 2 == 0) && (dst_width >= 0));
468
4.26M
  for (x = 0; x < src_width; ++x) {
469
3.64M
    dst_ptr[2 * x + 0] = (src_ptr[x + 0] * 3 + src_ptr[x + 1] * 1 + 2) >> 2;
470
3.64M
    dst_ptr[2 * x + 1] = (src_ptr[x + 0] * 1 + src_ptr[x + 1] * 3 + 2) >> 2;
471
3.64M
  }
472
627k
}
473
474
// Only suitable for at most 12bit range.
475
void ScaleRowUp2_Bilinear_16_C(const uint16_t* src_ptr,
476
                               ptrdiff_t src_stride,
477
                               uint16_t* dst_ptr,
478
                               ptrdiff_t dst_stride,
479
8.21k
                               int dst_width) {
480
8.21k
  const uint16_t* s = src_ptr;
481
8.21k
  const uint16_t* t = src_ptr + src_stride;
482
8.21k
  uint16_t* d = dst_ptr;
483
8.21k
  uint16_t* e = dst_ptr + dst_stride;
484
8.21k
  int src_width = dst_width >> 1;
485
8.21k
  int x;
486
8.21k
  assert((dst_width % 2 == 0) && (dst_width >= 0));
487
54.3k
  for (x = 0; x < src_width; ++x) {
488
46.1k
    d[2 * x + 0] =
489
46.1k
        (s[x + 0] * 9 + s[x + 1] * 3 + t[x + 0] * 3 + t[x + 1] * 1 + 8) >> 4;
490
46.1k
    d[2 * x + 1] =
491
46.1k
        (s[x + 0] * 3 + s[x + 1] * 9 + t[x + 0] * 1 + t[x + 1] * 3 + 8) >> 4;
492
46.1k
    e[2 * x + 0] =
493
46.1k
        (s[x + 0] * 3 + s[x + 1] * 1 + t[x + 0] * 9 + t[x + 1] * 3 + 8) >> 4;
494
46.1k
    e[2 * x + 1] =
495
46.1k
        (s[x + 0] * 1 + s[x + 1] * 3 + t[x + 0] * 3 + t[x + 1] * 9 + 8) >> 4;
496
46.1k
  }
497
8.21k
}
498
499
// Scales a single row of pixels using point sampling.
500
void ScaleCols_C(uint8_t* dst_ptr,
501
                 const uint8_t* src_ptr,
502
                 int dst_width,
503
                 int x,
504
597k
                 int dx) {
505
597k
  int j;
506
77.1M
  for (j = 0; j < dst_width - 1; j += 2) {
507
76.5M
    dst_ptr[0] = src_ptr[x >> 16];
508
76.5M
    x += dx;
509
76.5M
    dst_ptr[1] = src_ptr[x >> 16];
510
76.5M
    x += dx;
511
76.5M
    dst_ptr += 2;
512
76.5M
  }
513
597k
  if (dst_width & 1) {
514
514k
    dst_ptr[0] = src_ptr[x >> 16];
515
514k
  }
516
597k
}
517
518
void ScaleCols_16_C(uint16_t* dst_ptr,
519
                    const uint16_t* src_ptr,
520
                    int dst_width,
521
                    int x,
522
750k
                    int dx) {
523
750k
  int j;
524
164M
  for (j = 0; j < dst_width - 1; j += 2) {
525
163M
    dst_ptr[0] = src_ptr[x >> 16];
526
163M
    x += dx;
527
163M
    dst_ptr[1] = src_ptr[x >> 16];
528
163M
    x += dx;
529
163M
    dst_ptr += 2;
530
163M
  }
531
750k
  if (dst_width & 1) {
532
667k
    dst_ptr[0] = src_ptr[x >> 16];
533
667k
  }
534
750k
}
535
536
// Scales a single row of pixels up by 2x using point sampling.
537
void ScaleColsUp2_C(uint8_t* dst_ptr,
538
                    const uint8_t* src_ptr,
539
                    int dst_width,
540
                    int x,
541
182k
                    int dx) {
542
182k
  int j;
543
182k
  (void)x;
544
182k
  (void)dx;
545
365k
  for (j = 0; j < dst_width - 1; j += 2) {
546
182k
    dst_ptr[1] = dst_ptr[0] = src_ptr[0];
547
182k
    src_ptr += 1;
548
182k
    dst_ptr += 2;
549
182k
  }
550
182k
  if (dst_width & 1) {
551
0
    dst_ptr[0] = src_ptr[0];
552
0
  }
553
182k
}
554
555
void ScaleColsUp2_16_C(uint16_t* dst_ptr,
556
                       const uint16_t* src_ptr,
557
                       int dst_width,
558
                       int x,
559
148k
                       int dx) {
560
148k
  int j;
561
148k
  (void)x;
562
148k
  (void)dx;
563
296k
  for (j = 0; j < dst_width - 1; j += 2) {
564
148k
    dst_ptr[1] = dst_ptr[0] = src_ptr[0];
565
148k
    src_ptr += 1;
566
148k
    dst_ptr += 2;
567
148k
  }
568
148k
  if (dst_width & 1) {
569
0
    dst_ptr[0] = src_ptr[0];
570
0
  }
571
148k
}
572
573
// (1-f)a + fb can be replaced with a + f(b-a)
574
#if defined(__arm__) || defined(__aarch64__)
575
#define BLENDER(a, b, f) \
576
  (uint8_t)((int)(a) + ((((int)((f)) * ((int)(b) - (int)(a))) + 0x8000) >> 16))
577
#else
578
// Intel uses 7 bit math with rounding.
579
#define BLENDER(a, b, f) \
580
0
  (uint8_t)((int)(a) + (((int)((f) >> 9) * ((int)(b) - (int)(a)) + 0x40) >> 7))
581
#endif
582
583
void ScaleFilterCols_C(uint8_t* dst_ptr,
584
                       const uint8_t* src_ptr,
585
                       int dst_width,
586
                       int x,
587
0
                       int dx) {
588
0
  int j;
589
0
  for (j = 0; j < dst_width - 1; j += 2) {
590
0
    int xi = x >> 16;
591
0
    int a = src_ptr[xi];
592
0
    int b = src_ptr[xi + 1];
593
0
    dst_ptr[0] = BLENDER(a, b, x & 0xffff);
594
0
    x += dx;
595
0
    xi = x >> 16;
596
0
    a = src_ptr[xi];
597
0
    b = src_ptr[xi + 1];
598
0
    dst_ptr[1] = BLENDER(a, b, x & 0xffff);
599
0
    x += dx;
600
0
    dst_ptr += 2;
601
0
  }
602
0
  if (dst_width & 1) {
603
0
    int xi = x >> 16;
604
0
    int a = src_ptr[xi];
605
0
    int b = src_ptr[xi + 1];
606
0
    dst_ptr[0] = BLENDER(a, b, x & 0xffff);
607
0
  }
608
0
}
609
610
void ScaleFilterCols64_C(uint8_t* dst_ptr,
611
                         const uint8_t* src_ptr,
612
                         int dst_width,
613
                         int x32,
614
0
                         int dx) {
615
0
  int64_t x = (int64_t)(x32);
616
0
  int j;
617
0
  for (j = 0; j < dst_width - 1; j += 2) {
618
0
    int64_t xi = x >> 16;
619
0
    int a = src_ptr[xi];
620
0
    int b = src_ptr[xi + 1];
621
0
    dst_ptr[0] = BLENDER(a, b, x & 0xffff);
622
0
    x += dx;
623
0
    xi = x >> 16;
624
0
    a = src_ptr[xi];
625
0
    b = src_ptr[xi + 1];
626
0
    dst_ptr[1] = BLENDER(a, b, x & 0xffff);
627
0
    x += dx;
628
0
    dst_ptr += 2;
629
0
  }
630
0
  if (dst_width & 1) {
631
0
    int64_t xi = x >> 16;
632
0
    int a = src_ptr[xi];
633
0
    int b = src_ptr[xi + 1];
634
0
    dst_ptr[0] = BLENDER(a, b, x & 0xffff);
635
0
  }
636
0
}
637
#undef BLENDER
638
639
// Same as 8 bit arm blender but return is cast to uint16_t
640
#define BLENDER(a, b, f)                                                      \
641
258M
  (uint16_t)((int)(a) +                                                       \
642
258M
             (int)((((int64_t)((f)) * ((int64_t)(b) - (int)(a))) + 0x8000) >> \
643
258M
                   16))
644
645
void ScaleFilterCols_16_C(uint16_t* dst_ptr,
646
                          const uint16_t* src_ptr,
647
                          int dst_width,
648
                          int x,
649
634k
                          int dx) {
650
634k
  int j;
651
129M
  for (j = 0; j < dst_width - 1; j += 2) {
652
128M
    int xi = x >> 16;
653
128M
    int a = src_ptr[xi];
654
128M
    int b = src_ptr[xi + 1];
655
128M
    dst_ptr[0] = BLENDER(a, b, x & 0xffff);
656
128M
    x += dx;
657
128M
    xi = x >> 16;
658
128M
    a = src_ptr[xi];
659
128M
    b = src_ptr[xi + 1];
660
128M
    dst_ptr[1] = BLENDER(a, b, x & 0xffff);
661
128M
    x += dx;
662
128M
    dst_ptr += 2;
663
128M
  }
664
634k
  if (dst_width & 1) {
665
193k
    int xi = x >> 16;
666
193k
    int a = src_ptr[xi];
667
193k
    int b = src_ptr[xi + 1];
668
193k
    dst_ptr[0] = BLENDER(a, b, x & 0xffff);
669
193k
  }
670
634k
}
671
672
void ScaleFilterCols64_16_C(uint16_t* dst_ptr,
673
                            const uint16_t* src_ptr,
674
                            int dst_width,
675
                            int x32,
676
0
                            int dx) {
677
0
  int64_t x = (int64_t)(x32);
678
0
  int j;
679
0
  for (j = 0; j < dst_width - 1; j += 2) {
680
0
    int64_t xi = x >> 16;
681
0
    int a = src_ptr[xi];
682
0
    int b = src_ptr[xi + 1];
683
0
    dst_ptr[0] = BLENDER(a, b, x & 0xffff);
684
0
    x += dx;
685
0
    xi = x >> 16;
686
0
    a = src_ptr[xi];
687
0
    b = src_ptr[xi + 1];
688
0
    dst_ptr[1] = BLENDER(a, b, x & 0xffff);
689
0
    x += dx;
690
0
    dst_ptr += 2;
691
0
  }
692
0
  if (dst_width & 1) {
693
0
    int64_t xi = x >> 16;
694
0
    int a = src_ptr[xi];
695
0
    int b = src_ptr[xi + 1];
696
0
    dst_ptr[0] = BLENDER(a, b, x & 0xffff);
697
0
  }
698
0
}
699
#undef BLENDER
700
701
void ScaleRowDown38_C(const uint8_t* src_ptr,
702
                      ptrdiff_t src_stride,
703
                      uint8_t* dst,
704
0
                      int dst_width) {
705
0
  int x;
706
0
  (void)src_stride;
707
0
  assert(dst_width % 3 == 0);
708
0
  for (x = 0; x < dst_width; x += 3) {
709
0
    dst[0] = src_ptr[0];
710
0
    dst[1] = src_ptr[3];
711
0
    dst[2] = src_ptr[6];
712
0
    dst += 3;
713
0
    src_ptr += 8;
714
0
  }
715
0
}
716
717
void ScaleRowDown38_16_C(const uint16_t* src_ptr,
718
                         ptrdiff_t src_stride,
719
                         uint16_t* dst,
720
0
                         int dst_width) {
721
0
  int x;
722
0
  (void)src_stride;
723
0
  assert(dst_width % 3 == 0);
724
0
  for (x = 0; x < dst_width; x += 3) {
725
0
    dst[0] = src_ptr[0];
726
0
    dst[1] = src_ptr[3];
727
0
    dst[2] = src_ptr[6];
728
0
    dst += 3;
729
0
    src_ptr += 8;
730
0
  }
731
0
}
732
733
// 8x3 -> 3x1
734
void ScaleRowDown38_3_Box_C(const uint8_t* src_ptr,
735
                            ptrdiff_t src_stride,
736
                            uint8_t* dst_ptr,
737
0
                            int dst_width) {
738
0
  int i;
739
0
  assert((dst_width % 3 == 0) && (dst_width > 0));
740
0
  for (i = 0; i < dst_width; i += 3) {
741
0
    dst_ptr[0] = (src_ptr[0] + src_ptr[1] + src_ptr[2] +
742
0
                  src_ptr[src_stride + 0] + src_ptr[src_stride + 1] +
743
0
                  src_ptr[src_stride + 2] + src_ptr[src_stride * 2 + 0] +
744
0
                  src_ptr[src_stride * 2 + 1] + src_ptr[src_stride * 2 + 2]) *
745
0
                     (65536 / 9) >>
746
0
                 16;
747
0
    dst_ptr[1] = (src_ptr[3] + src_ptr[4] + src_ptr[5] +
748
0
                  src_ptr[src_stride + 3] + src_ptr[src_stride + 4] +
749
0
                  src_ptr[src_stride + 5] + src_ptr[src_stride * 2 + 3] +
750
0
                  src_ptr[src_stride * 2 + 4] + src_ptr[src_stride * 2 + 5]) *
751
0
                     (65536 / 9) >>
752
0
                 16;
753
0
    dst_ptr[2] = (src_ptr[6] + src_ptr[7] + src_ptr[src_stride + 6] +
754
0
                  src_ptr[src_stride + 7] + src_ptr[src_stride * 2 + 6] +
755
0
                  src_ptr[src_stride * 2 + 7]) *
756
0
                     (65536 / 6) >>
757
0
                 16;
758
0
    src_ptr += 8;
759
0
    dst_ptr += 3;
760
0
  }
761
0
}
762
763
void ScaleRowDown38_3_Box_16_C(const uint16_t* src_ptr,
764
                               ptrdiff_t src_stride,
765
                               uint16_t* dst_ptr,
766
48
                               int dst_width) {
767
48
  int i;
768
48
  assert((dst_width % 3 == 0) && (dst_width > 0));
769
240
  for (i = 0; i < dst_width; i += 3) {
770
192
    dst_ptr[0] = (src_ptr[0] + src_ptr[1] + src_ptr[2] +
771
192
                  src_ptr[src_stride + 0] + src_ptr[src_stride + 1] +
772
192
                  src_ptr[src_stride + 2] + src_ptr[src_stride * 2 + 0] +
773
192
                  src_ptr[src_stride * 2 + 1] + src_ptr[src_stride * 2 + 2]) *
774
192
                     (65536u / 9u) >>
775
192
                 16;
776
192
    dst_ptr[1] = (src_ptr[3] + src_ptr[4] + src_ptr[5] +
777
192
                  src_ptr[src_stride + 3] + src_ptr[src_stride + 4] +
778
192
                  src_ptr[src_stride + 5] + src_ptr[src_stride * 2 + 3] +
779
192
                  src_ptr[src_stride * 2 + 4] + src_ptr[src_stride * 2 + 5]) *
780
192
                     (65536u / 9u) >>
781
192
                 16;
782
192
    dst_ptr[2] = (src_ptr[6] + src_ptr[7] + src_ptr[src_stride + 6] +
783
192
                  src_ptr[src_stride + 7] + src_ptr[src_stride * 2 + 6] +
784
192
                  src_ptr[src_stride * 2 + 7]) *
785
192
                     (65536u / 6u) >>
786
192
                 16;
787
192
    src_ptr += 8;
788
192
    dst_ptr += 3;
789
192
  }
790
48
}
791
792
// 8x2 -> 3x1
793
void ScaleRowDown38_2_Box_C(const uint8_t* src_ptr,
794
                            ptrdiff_t src_stride,
795
                            uint8_t* dst_ptr,
796
0
                            int dst_width) {
797
0
  int i;
798
0
  assert((dst_width % 3 == 0) && (dst_width > 0));
799
0
  for (i = 0; i < dst_width; i += 3) {
800
0
    dst_ptr[0] =
801
0
        (src_ptr[0] + src_ptr[1] + src_ptr[2] + src_ptr[src_stride + 0] +
802
0
         src_ptr[src_stride + 1] + src_ptr[src_stride + 2]) *
803
0
            (65536 / 6) >>
804
0
        16;
805
0
    dst_ptr[1] =
806
0
        (src_ptr[3] + src_ptr[4] + src_ptr[5] + src_ptr[src_stride + 3] +
807
0
         src_ptr[src_stride + 4] + src_ptr[src_stride + 5]) *
808
0
            (65536 / 6) >>
809
0
        16;
810
0
    dst_ptr[2] = (src_ptr[6] + src_ptr[7] + src_ptr[src_stride + 6] +
811
0
                  src_ptr[src_stride + 7]) *
812
0
                     (65536 / 4) >>
813
0
                 16;
814
0
    src_ptr += 8;
815
0
    dst_ptr += 3;
816
0
  }
817
0
}
818
819
void ScaleRowDown38_2_Box_16_C(const uint16_t* src_ptr,
820
                               ptrdiff_t src_stride,
821
                               uint16_t* dst_ptr,
822
24
                               int dst_width) {
823
24
  int i;
824
24
  assert((dst_width % 3 == 0) && (dst_width > 0));
825
120
  for (i = 0; i < dst_width; i += 3) {
826
96
    dst_ptr[0] =
827
96
        (src_ptr[0] + src_ptr[1] + src_ptr[2] + src_ptr[src_stride + 0] +
828
96
         src_ptr[src_stride + 1] + src_ptr[src_stride + 2]) *
829
96
            (65536u / 6u) >>
830
96
        16;
831
96
    dst_ptr[1] =
832
96
        (src_ptr[3] + src_ptr[4] + src_ptr[5] + src_ptr[src_stride + 3] +
833
96
         src_ptr[src_stride + 4] + src_ptr[src_stride + 5]) *
834
96
            (65536u / 6u) >>
835
96
        16;
836
96
    dst_ptr[2] = (src_ptr[6] + src_ptr[7] + src_ptr[src_stride + 6] +
837
96
                  src_ptr[src_stride + 7]) *
838
96
                     (65536u / 4u) >>
839
96
                 16;
840
96
    src_ptr += 8;
841
96
    dst_ptr += 3;
842
96
  }
843
24
}
844
845
571k
void ScaleAddRow_C(const uint8_t* src_ptr, uint16_t* dst_ptr, int src_width) {
846
571k
  int x;
847
571k
  assert(src_width > 0);
848
5.49M
  for (x = 0; x < src_width - 1; x += 2) {
849
4.91M
    dst_ptr[0] += src_ptr[0];
850
4.91M
    dst_ptr[1] += src_ptr[1];
851
4.91M
    src_ptr += 2;
852
4.91M
    dst_ptr += 2;
853
4.91M
  }
854
571k
  if (src_width & 1) {
855
275k
    dst_ptr[0] += src_ptr[0];
856
275k
  }
857
571k
}
858
859
void ScaleAddRow_16_C(const uint16_t* src_ptr,
860
                      uint32_t* dst_ptr,
861
1.23M
                      int src_width) {
862
1.23M
  int x;
863
1.23M
  assert(src_width > 0);
864
101M
  for (x = 0; x < src_width - 1; x += 2) {
865
100M
    dst_ptr[0] += src_ptr[0];
866
100M
    dst_ptr[1] += src_ptr[1];
867
100M
    src_ptr += 2;
868
100M
    dst_ptr += 2;
869
100M
  }
870
1.23M
  if (src_width & 1) {
871
226k
    dst_ptr[0] += src_ptr[0];
872
226k
  }
873
1.23M
}
874
875
// ARGB scale row functions
876
877
void ScaleARGBRowDown2_C(const uint8_t* src_argb,
878
                         ptrdiff_t src_stride,
879
                         uint8_t* dst_argb,
880
0
                         int dst_width) {
881
0
  const uint32_t* src = (const uint32_t*)(src_argb);
882
0
  uint32_t* dst = (uint32_t*)(dst_argb);
883
0
  int x;
884
0
  (void)src_stride;
885
0
  for (x = 0; x < dst_width - 1; x += 2) {
886
0
    dst[0] = src[1];
887
0
    dst[1] = src[3];
888
0
    src += 4;
889
0
    dst += 2;
890
0
  }
891
0
  if (dst_width & 1) {
892
0
    dst[0] = src[1];
893
0
  }
894
0
}
895
896
void ScaleARGBRowDown2Linear_C(const uint8_t* src_argb,
897
                               ptrdiff_t src_stride,
898
                               uint8_t* dst_argb,
899
0
                               int dst_width) {
900
0
  int x;
901
0
  (void)src_stride;
902
0
  for (x = 0; x < dst_width; ++x) {
903
0
    dst_argb[0] = (src_argb[0] + src_argb[4] + 1) >> 1;
904
0
    dst_argb[1] = (src_argb[1] + src_argb[5] + 1) >> 1;
905
0
    dst_argb[2] = (src_argb[2] + src_argb[6] + 1) >> 1;
906
0
    dst_argb[3] = (src_argb[3] + src_argb[7] + 1) >> 1;
907
0
    src_argb += 8;
908
0
    dst_argb += 4;
909
0
  }
910
0
}
911
912
void ScaleARGBRowDown2Box_C(const uint8_t* src_argb,
913
                            ptrdiff_t src_stride,
914
                            uint8_t* dst_argb,
915
0
                            int dst_width) {
916
0
  int x;
917
0
  for (x = 0; x < dst_width; ++x) {
918
0
    dst_argb[0] = (src_argb[0] + src_argb[4] + src_argb[src_stride] +
919
0
                   src_argb[src_stride + 4] + 2) >>
920
0
                  2;
921
0
    dst_argb[1] = (src_argb[1] + src_argb[5] + src_argb[src_stride + 1] +
922
0
                   src_argb[src_stride + 5] + 2) >>
923
0
                  2;
924
0
    dst_argb[2] = (src_argb[2] + src_argb[6] + src_argb[src_stride + 2] +
925
0
                   src_argb[src_stride + 6] + 2) >>
926
0
                  2;
927
0
    dst_argb[3] = (src_argb[3] + src_argb[7] + src_argb[src_stride + 3] +
928
0
                   src_argb[src_stride + 7] + 2) >>
929
0
                  2;
930
0
    src_argb += 8;
931
0
    dst_argb += 4;
932
0
  }
933
0
}
934
935
void ScaleARGBRowDownEven_C(const uint8_t* src_argb,
936
                            ptrdiff_t src_stride,
937
                            int src_stepx,
938
                            uint8_t* dst_argb,
939
0
                            int dst_width) {
940
0
  const uint32_t* src = (const uint32_t*)(src_argb);
941
0
  uint32_t* dst = (uint32_t*)(dst_argb);
942
0
  (void)src_stride;
943
0
  int x;
944
0
  for (x = 0; x < dst_width - 1; x += 2) {
945
0
    dst[0] = src[0];
946
0
    dst[1] = src[src_stepx];
947
0
    src += src_stepx * 2;
948
0
    dst += 2;
949
0
  }
950
0
  if (dst_width & 1) {
951
0
    dst[0] = src[0];
952
0
  }
953
0
}
954
955
void ScaleARGBRowDownEvenBox_C(const uint8_t* src_argb,
956
                               ptrdiff_t src_stride,
957
                               int src_stepx,
958
                               uint8_t* dst_argb,
959
0
                               int dst_width) {
960
0
  int x;
961
0
  for (x = 0; x < dst_width; ++x) {
962
0
    dst_argb[0] = (src_argb[0] + src_argb[4] + src_argb[src_stride] +
963
0
                   src_argb[src_stride + 4] + 2) >>
964
0
                  2;
965
0
    dst_argb[1] = (src_argb[1] + src_argb[5] + src_argb[src_stride + 1] +
966
0
                   src_argb[src_stride + 5] + 2) >>
967
0
                  2;
968
0
    dst_argb[2] = (src_argb[2] + src_argb[6] + src_argb[src_stride + 2] +
969
0
                   src_argb[src_stride + 6] + 2) >>
970
0
                  2;
971
0
    dst_argb[3] = (src_argb[3] + src_argb[7] + src_argb[src_stride + 3] +
972
0
                   src_argb[src_stride + 7] + 2) >>
973
0
                  2;
974
0
    src_argb += src_stepx * 4;
975
0
    dst_argb += 4;
976
0
  }
977
0
}
978
979
// Scales a single row of pixels using point sampling.
980
void ScaleARGBCols_C(uint8_t* dst_argb,
981
                     const uint8_t* src_argb,
982
                     int dst_width,
983
                     int x,
984
0
                     int dx) {
985
0
  const uint32_t* src = (const uint32_t*)(src_argb);
986
0
  uint32_t* dst = (uint32_t*)(dst_argb);
987
0
  int j;
988
0
  for (j = 0; j < dst_width - 1; j += 2) {
989
0
    dst[0] = src[x >> 16];
990
0
    x += dx;
991
0
    dst[1] = src[x >> 16];
992
0
    x += dx;
993
0
    dst += 2;
994
0
  }
995
0
  if (dst_width & 1) {
996
0
    dst[0] = src[x >> 16];
997
0
  }
998
0
}
999
1000
void ScaleARGBCols64_C(uint8_t* dst_argb,
1001
                       const uint8_t* src_argb,
1002
                       int dst_width,
1003
                       int x32,
1004
0
                       int dx) {
1005
0
  int64_t x = (int64_t)(x32);
1006
0
  const uint32_t* src = (const uint32_t*)(src_argb);
1007
0
  uint32_t* dst = (uint32_t*)(dst_argb);
1008
0
  int j;
1009
0
  for (j = 0; j < dst_width - 1; j += 2) {
1010
0
    dst[0] = src[x >> 16];
1011
0
    x += dx;
1012
0
    dst[1] = src[x >> 16];
1013
0
    x += dx;
1014
0
    dst += 2;
1015
0
  }
1016
0
  if (dst_width & 1) {
1017
0
    dst[0] = src[x >> 16];
1018
0
  }
1019
0
}
1020
1021
// Scales a single row of pixels up by 2x using point sampling.
1022
void ScaleARGBColsUp2_C(uint8_t* dst_argb,
1023
                        const uint8_t* src_argb,
1024
                        int dst_width,
1025
                        int x,
1026
0
                        int dx) {
1027
0
  const uint32_t* src = (const uint32_t*)(src_argb);
1028
0
  uint32_t* dst = (uint32_t*)(dst_argb);
1029
0
  int j;
1030
0
  (void)x;
1031
0
  (void)dx;
1032
0
  for (j = 0; j < dst_width - 1; j += 2) {
1033
0
    dst[1] = dst[0] = src[0];
1034
0
    src += 1;
1035
0
    dst += 2;
1036
0
  }
1037
0
  if (dst_width & 1) {
1038
0
    dst[0] = src[0];
1039
0
  }
1040
0
}
1041
1042
// TODO(fbarchard): Replace 0x7f ^ f with 128-f.  bug=607.
1043
// Mimics SSSE3 blender
1044
0
#define BLENDER1(a, b, f) ((a) * (0x7f ^ f) + (b) * f) >> 7
1045
#define BLENDERC(a, b, f, s) \
1046
0
  (uint32_t)(BLENDER1(((a) >> s) & 255, ((b) >> s) & 255, f) << s)
1047
#define BLENDER(a, b, f)                                                 \
1048
0
  BLENDERC(a, b, f, 24) | BLENDERC(a, b, f, 16) | BLENDERC(a, b, f, 8) | \
1049
0
      BLENDERC(a, b, f, 0)
1050
1051
void ScaleARGBFilterCols_C(uint8_t* dst_argb,
1052
                           const uint8_t* src_argb,
1053
                           int dst_width,
1054
                           int x,
1055
0
                           int dx) {
1056
0
  const uint32_t* src = (const uint32_t*)(src_argb);
1057
0
  uint32_t* dst = (uint32_t*)(dst_argb);
1058
0
  int j;
1059
0
  for (j = 0; j < dst_width - 1; j += 2) {
1060
0
    int xi = x >> 16;
1061
0
    int xf = (x >> 9) & 0x7f;
1062
0
    uint32_t a = src[xi];
1063
0
    uint32_t b = src[xi + 1];
1064
0
    dst[0] = BLENDER(a, b, xf);
1065
0
    x += dx;
1066
0
    xi = x >> 16;
1067
0
    xf = (x >> 9) & 0x7f;
1068
0
    a = src[xi];
1069
0
    b = src[xi + 1];
1070
0
    dst[1] = BLENDER(a, b, xf);
1071
0
    x += dx;
1072
0
    dst += 2;
1073
0
  }
1074
0
  if (dst_width & 1) {
1075
0
    int xi = x >> 16;
1076
0
    int xf = (x >> 9) & 0x7f;
1077
0
    uint32_t a = src[xi];
1078
0
    uint32_t b = src[xi + 1];
1079
0
    dst[0] = BLENDER(a, b, xf);
1080
0
  }
1081
0
}
1082
1083
void ScaleARGBFilterCols64_C(uint8_t* dst_argb,
1084
                             const uint8_t* src_argb,
1085
                             int dst_width,
1086
                             int x32,
1087
0
                             int dx) {
1088
0
  int64_t x = (int64_t)(x32);
1089
0
  const uint32_t* src = (const uint32_t*)(src_argb);
1090
0
  uint32_t* dst = (uint32_t*)(dst_argb);
1091
0
  int j;
1092
0
  for (j = 0; j < dst_width - 1; j += 2) {
1093
0
    int64_t xi = x >> 16;
1094
0
    int xf = (x >> 9) & 0x7f;
1095
0
    uint32_t a = src[xi];
1096
0
    uint32_t b = src[xi + 1];
1097
0
    dst[0] = BLENDER(a, b, xf);
1098
0
    x += dx;
1099
0
    xi = x >> 16;
1100
0
    xf = (x >> 9) & 0x7f;
1101
0
    a = src[xi];
1102
0
    b = src[xi + 1];
1103
0
    dst[1] = BLENDER(a, b, xf);
1104
0
    x += dx;
1105
0
    dst += 2;
1106
0
  }
1107
0
  if (dst_width & 1) {
1108
0
    int64_t xi = x >> 16;
1109
0
    int xf = (x >> 9) & 0x7f;
1110
0
    uint32_t a = src[xi];
1111
0
    uint32_t b = src[xi + 1];
1112
0
    dst[0] = BLENDER(a, b, xf);
1113
0
  }
1114
0
}
1115
#undef BLENDER1
1116
#undef BLENDERC
1117
#undef BLENDER
1118
1119
// UV scale row functions
1120
// same as ARGB but 2 channels
1121
1122
void ScaleUVRowDown2_C(const uint8_t* src_uv,
1123
                       ptrdiff_t src_stride,
1124
                       uint8_t* dst_uv,
1125
0
                       int dst_width) {
1126
0
  int x;
1127
0
  (void)src_stride;
1128
0
  for (x = 0; x < dst_width; ++x) {
1129
0
    dst_uv[0] = src_uv[2];  // Store the 2nd UV
1130
0
    dst_uv[1] = src_uv[3];
1131
0
    src_uv += 4;
1132
0
    dst_uv += 2;
1133
0
  }
1134
0
}
1135
1136
void ScaleUVRowDown2Linear_C(const uint8_t* src_uv,
1137
                             ptrdiff_t src_stride,
1138
                             uint8_t* dst_uv,
1139
0
                             int dst_width) {
1140
0
  int x;
1141
0
  (void)src_stride;
1142
0
  for (x = 0; x < dst_width; ++x) {
1143
0
    dst_uv[0] = (src_uv[0] + src_uv[2] + 1) >> 1;
1144
0
    dst_uv[1] = (src_uv[1] + src_uv[3] + 1) >> 1;
1145
0
    src_uv += 4;
1146
0
    dst_uv += 2;
1147
0
  }
1148
0
}
1149
1150
void ScaleUVRowDown2Box_C(const uint8_t* src_uv,
1151
                          ptrdiff_t src_stride,
1152
                          uint8_t* dst_uv,
1153
0
                          int dst_width) {
1154
0
  int x;
1155
0
  for (x = 0; x < dst_width; ++x) {
1156
0
    dst_uv[0] = (src_uv[0] + src_uv[2] + src_uv[src_stride] +
1157
0
                 src_uv[src_stride + 2] + 2) >>
1158
0
                2;
1159
0
    dst_uv[1] = (src_uv[1] + src_uv[3] + src_uv[src_stride + 1] +
1160
0
                 src_uv[src_stride + 3] + 2) >>
1161
0
                2;
1162
0
    src_uv += 4;
1163
0
    dst_uv += 2;
1164
0
  }
1165
0
}
1166
1167
void ScaleUVRowDownEven_C(const uint8_t* src_uv,
1168
                          ptrdiff_t src_stride,
1169
                          int src_stepx,
1170
                          uint8_t* dst_uv,
1171
0
                          int dst_width) {
1172
0
  const uint16_t* src = (const uint16_t*)(src_uv);
1173
0
  uint16_t* dst = (uint16_t*)(dst_uv);
1174
0
  (void)src_stride;
1175
0
  int x;
1176
0
  for (x = 0; x < dst_width - 1; x += 2) {
1177
0
    dst[0] = src[0];
1178
0
    dst[1] = src[src_stepx];
1179
0
    src += src_stepx * 2;
1180
0
    dst += 2;
1181
0
  }
1182
0
  if (dst_width & 1) {
1183
0
    dst[0] = src[0];
1184
0
  }
1185
0
}
1186
1187
void ScaleUVRowDownEvenBox_C(const uint8_t* src_uv,
1188
                             ptrdiff_t src_stride,
1189
                             int src_stepx,
1190
                             uint8_t* dst_uv,
1191
0
                             int dst_width) {
1192
0
  int x;
1193
0
  for (x = 0; x < dst_width; ++x) {
1194
0
    dst_uv[0] = (src_uv[0] + src_uv[2] + src_uv[src_stride] +
1195
0
                 src_uv[src_stride + 2] + 2) >>
1196
0
                2;
1197
0
    dst_uv[1] = (src_uv[1] + src_uv[3] + src_uv[src_stride + 1] +
1198
0
                 src_uv[src_stride + 3] + 2) >>
1199
0
                2;
1200
0
    src_uv += src_stepx * 2;
1201
0
    dst_uv += 2;
1202
0
  }
1203
0
}
1204
1205
void ScaleUVRowUp2_Linear_C(const uint8_t* src_ptr,
1206
                            uint8_t* dst_ptr,
1207
0
                            int dst_width) {
1208
0
  int src_width = dst_width >> 1;
1209
0
  int x;
1210
0
  assert((dst_width % 2 == 0) && (dst_width >= 0));
1211
0
  for (x = 0; x < src_width; ++x) {
1212
0
    dst_ptr[4 * x + 0] =
1213
0
        (src_ptr[2 * x + 0] * 3 + src_ptr[2 * x + 2] * 1 + 2) >> 2;
1214
0
    dst_ptr[4 * x + 1] =
1215
0
        (src_ptr[2 * x + 1] * 3 + src_ptr[2 * x + 3] * 1 + 2) >> 2;
1216
0
    dst_ptr[4 * x + 2] =
1217
0
        (src_ptr[2 * x + 0] * 1 + src_ptr[2 * x + 2] * 3 + 2) >> 2;
1218
0
    dst_ptr[4 * x + 3] =
1219
0
        (src_ptr[2 * x + 1] * 1 + src_ptr[2 * x + 3] * 3 + 2) >> 2;
1220
0
  }
1221
0
}
1222
1223
void ScaleUVRowUp2_Bilinear_C(const uint8_t* src_ptr,
1224
                              ptrdiff_t src_stride,
1225
                              uint8_t* dst_ptr,
1226
                              ptrdiff_t dst_stride,
1227
0
                              int dst_width) {
1228
0
  const uint8_t* s = src_ptr;
1229
0
  const uint8_t* t = src_ptr + src_stride;
1230
0
  uint8_t* d = dst_ptr;
1231
0
  uint8_t* e = dst_ptr + dst_stride;
1232
0
  int src_width = dst_width >> 1;
1233
0
  int x;
1234
0
  assert((dst_width % 2 == 0) && (dst_width >= 0));
1235
0
  for (x = 0; x < src_width; ++x) {
1236
0
    d[4 * x + 0] = (s[2 * x + 0] * 9 + s[2 * x + 2] * 3 + t[2 * x + 0] * 3 +
1237
0
                    t[2 * x + 2] * 1 + 8) >>
1238
0
                   4;
1239
0
    d[4 * x + 1] = (s[2 * x + 1] * 9 + s[2 * x + 3] * 3 + t[2 * x + 1] * 3 +
1240
0
                    t[2 * x + 3] * 1 + 8) >>
1241
0
                   4;
1242
0
    d[4 * x + 2] = (s[2 * x + 0] * 3 + s[2 * x + 2] * 9 + t[2 * x + 0] * 1 +
1243
0
                    t[2 * x + 2] * 3 + 8) >>
1244
0
                   4;
1245
0
    d[4 * x + 3] = (s[2 * x + 1] * 3 + s[2 * x + 3] * 9 + t[2 * x + 1] * 1 +
1246
0
                    t[2 * x + 3] * 3 + 8) >>
1247
0
                   4;
1248
0
    e[4 * x + 0] = (s[2 * x + 0] * 3 + s[2 * x + 2] * 1 + t[2 * x + 0] * 9 +
1249
0
                    t[2 * x + 2] * 3 + 8) >>
1250
0
                   4;
1251
0
    e[4 * x + 1] = (s[2 * x + 1] * 3 + s[2 * x + 3] * 1 + t[2 * x + 1] * 9 +
1252
0
                    t[2 * x + 3] * 3 + 8) >>
1253
0
                   4;
1254
0
    e[4 * x + 2] = (s[2 * x + 0] * 1 + s[2 * x + 2] * 3 + t[2 * x + 0] * 3 +
1255
0
                    t[2 * x + 2] * 9 + 8) >>
1256
0
                   4;
1257
0
    e[4 * x + 3] = (s[2 * x + 1] * 1 + s[2 * x + 3] * 3 + t[2 * x + 1] * 3 +
1258
0
                    t[2 * x + 3] * 9 + 8) >>
1259
0
                   4;
1260
0
  }
1261
0
}
1262
1263
void ScaleUVRowUp2_Linear_16_C(const uint16_t* src_ptr,
1264
                               uint16_t* dst_ptr,
1265
0
                               int dst_width) {
1266
0
  int src_width = dst_width >> 1;
1267
0
  int x;
1268
0
  assert((dst_width % 2 == 0) && (dst_width >= 0));
1269
0
  for (x = 0; x < src_width; ++x) {
1270
0
    dst_ptr[4 * x + 0] =
1271
0
        (src_ptr[2 * x + 0] * 3 + src_ptr[2 * x + 2] * 1 + 2) >> 2;
1272
0
    dst_ptr[4 * x + 1] =
1273
0
        (src_ptr[2 * x + 1] * 3 + src_ptr[2 * x + 3] * 1 + 2) >> 2;
1274
0
    dst_ptr[4 * x + 2] =
1275
0
        (src_ptr[2 * x + 0] * 1 + src_ptr[2 * x + 2] * 3 + 2) >> 2;
1276
0
    dst_ptr[4 * x + 3] =
1277
0
        (src_ptr[2 * x + 1] * 1 + src_ptr[2 * x + 3] * 3 + 2) >> 2;
1278
0
  }
1279
0
}
1280
1281
void ScaleUVRowUp2_Bilinear_16_C(const uint16_t* src_ptr,
1282
                                 ptrdiff_t src_stride,
1283
                                 uint16_t* dst_ptr,
1284
                                 ptrdiff_t dst_stride,
1285
0
                                 int dst_width) {
1286
0
  const uint16_t* s = src_ptr;
1287
0
  const uint16_t* t = src_ptr + src_stride;
1288
0
  uint16_t* d = dst_ptr;
1289
0
  uint16_t* e = dst_ptr + dst_stride;
1290
0
  int src_width = dst_width >> 1;
1291
0
  int x;
1292
0
  assert((dst_width % 2 == 0) && (dst_width >= 0));
1293
0
  for (x = 0; x < src_width; ++x) {
1294
0
    d[4 * x + 0] = (s[2 * x + 0] * 9 + s[2 * x + 2] * 3 + t[2 * x + 0] * 3 +
1295
0
                    t[2 * x + 2] * 1 + 8) >>
1296
0
                   4;
1297
0
    d[4 * x + 1] = (s[2 * x + 1] * 9 + s[2 * x + 3] * 3 + t[2 * x + 1] * 3 +
1298
0
                    t[2 * x + 3] * 1 + 8) >>
1299
0
                   4;
1300
0
    d[4 * x + 2] = (s[2 * x + 0] * 3 + s[2 * x + 2] * 9 + t[2 * x + 0] * 1 +
1301
0
                    t[2 * x + 2] * 3 + 8) >>
1302
0
                   4;
1303
0
    d[4 * x + 3] = (s[2 * x + 1] * 3 + s[2 * x + 3] * 9 + t[2 * x + 1] * 1 +
1304
0
                    t[2 * x + 3] * 3 + 8) >>
1305
0
                   4;
1306
0
    e[4 * x + 0] = (s[2 * x + 0] * 3 + s[2 * x + 2] * 1 + t[2 * x + 0] * 9 +
1307
0
                    t[2 * x + 2] * 3 + 8) >>
1308
0
                   4;
1309
0
    e[4 * x + 1] = (s[2 * x + 1] * 3 + s[2 * x + 3] * 1 + t[2 * x + 1] * 9 +
1310
0
                    t[2 * x + 3] * 3 + 8) >>
1311
0
                   4;
1312
0
    e[4 * x + 2] = (s[2 * x + 0] * 1 + s[2 * x + 2] * 3 + t[2 * x + 0] * 3 +
1313
0
                    t[2 * x + 2] * 9 + 8) >>
1314
0
                   4;
1315
0
    e[4 * x + 3] = (s[2 * x + 1] * 1 + s[2 * x + 3] * 3 + t[2 * x + 1] * 3 +
1316
0
                    t[2 * x + 3] * 9 + 8) >>
1317
0
                   4;
1318
0
  }
1319
0
}
1320
1321
// Scales a single row of pixels using point sampling.
1322
void ScaleUVCols_C(uint8_t* dst_uv,
1323
                   const uint8_t* src_uv,
1324
                   int dst_width,
1325
                   int x,
1326
0
                   int dx) {
1327
0
  const uint16_t* src = (const uint16_t*)(src_uv);
1328
0
  uint16_t* dst = (uint16_t*)(dst_uv);
1329
0
  int j;
1330
0
  for (j = 0; j < dst_width - 1; j += 2) {
1331
0
    dst[0] = src[x >> 16];
1332
0
    x += dx;
1333
0
    dst[1] = src[x >> 16];
1334
0
    x += dx;
1335
0
    dst += 2;
1336
0
  }
1337
0
  if (dst_width & 1) {
1338
0
    dst[0] = src[x >> 16];
1339
0
  }
1340
0
}
1341
1342
void ScaleUVCols64_C(uint8_t* dst_uv,
1343
                     const uint8_t* src_uv,
1344
                     int dst_width,
1345
                     int x32,
1346
0
                     int dx) {
1347
0
  int64_t x = (int64_t)(x32);
1348
0
  const uint16_t* src = (const uint16_t*)(src_uv);
1349
0
  uint16_t* dst = (uint16_t*)(dst_uv);
1350
0
  int j;
1351
0
  for (j = 0; j < dst_width - 1; j += 2) {
1352
0
    dst[0] = src[x >> 16];
1353
0
    x += dx;
1354
0
    dst[1] = src[x >> 16];
1355
0
    x += dx;
1356
0
    dst += 2;
1357
0
  }
1358
0
  if (dst_width & 1) {
1359
0
    dst[0] = src[x >> 16];
1360
0
  }
1361
0
}
1362
1363
// Scales a single row of pixels up by 2x using point sampling.
1364
void ScaleUVColsUp2_C(uint8_t* dst_uv,
1365
                      const uint8_t* src_uv,
1366
                      int dst_width,
1367
                      int x,
1368
0
                      int dx) {
1369
0
  const uint16_t* src = (const uint16_t*)(src_uv);
1370
0
  uint16_t* dst = (uint16_t*)(dst_uv);
1371
0
  int j;
1372
0
  (void)x;
1373
0
  (void)dx;
1374
0
  for (j = 0; j < dst_width - 1; j += 2) {
1375
0
    dst[1] = dst[0] = src[0];
1376
0
    src += 1;
1377
0
    dst += 2;
1378
0
  }
1379
0
  if (dst_width & 1) {
1380
0
    dst[0] = src[0];
1381
0
  }
1382
0
}
1383
1384
// Performs (a + ((f * (b - a) + 64) >> 7)) which is equivalent of
1385
// ((a * (128 - f) + b * f + 64) >> 7).
1386
0
#define BLENDER1(a, b, f) ((a) + (((f) * ((b) - (a)) + 64) >> 7))
1387
#define BLENDERC(a, b, f, s) \
1388
0
  (uint16_t)(BLENDER1(((a) >> s) & 255, ((b) >> s) & 255, f) << s)
1389
0
#define BLENDER(a, b, f) BLENDERC(a, b, f, 8) | BLENDERC(a, b, f, 0)
1390
1391
void ScaleUVFilterCols_C(uint8_t* dst_uv,
1392
                         const uint8_t* src_uv,
1393
                         int dst_width,
1394
                         int x,
1395
0
                         int dx) {
1396
0
  const uint16_t* src = (const uint16_t*)(src_uv);
1397
0
  uint16_t* dst = (uint16_t*)(dst_uv);
1398
0
  int j;
1399
0
  for (j = 0; j < dst_width - 1; j += 2) {
1400
0
    int xi = x >> 16;
1401
0
    int xf = (x >> 9) & 0x7f;
1402
0
    uint16_t a = src[xi];
1403
0
    uint16_t b = src[xi + 1];
1404
0
    dst[0] = BLENDER(a, b, xf);
1405
0
    x += dx;
1406
0
    xi = x >> 16;
1407
0
    xf = (x >> 9) & 0x7f;
1408
0
    a = src[xi];
1409
0
    b = src[xi + 1];
1410
0
    dst[1] = BLENDER(a, b, xf);
1411
0
    x += dx;
1412
0
    dst += 2;
1413
0
  }
1414
0
  if (dst_width & 1) {
1415
0
    int xi = x >> 16;
1416
0
    int xf = (x >> 9) & 0x7f;
1417
0
    uint16_t a = src[xi];
1418
0
    uint16_t b = src[xi + 1];
1419
0
    dst[0] = BLENDER(a, b, xf);
1420
0
  }
1421
0
}
1422
1423
void ScaleUVFilterCols64_C(uint8_t* dst_uv,
1424
                           const uint8_t* src_uv,
1425
                           int dst_width,
1426
                           int x32,
1427
0
                           int dx) {
1428
0
  int64_t x = (int64_t)(x32);
1429
0
  const uint16_t* src = (const uint16_t*)(src_uv);
1430
0
  uint16_t* dst = (uint16_t*)(dst_uv);
1431
0
  int j;
1432
0
  for (j = 0; j < dst_width - 1; j += 2) {
1433
0
    int64_t xi = x >> 16;
1434
0
    int xf = (x >> 9) & 0x7f;
1435
0
    uint16_t a = src[xi];
1436
0
    uint16_t b = src[xi + 1];
1437
0
    dst[0] = BLENDER(a, b, xf);
1438
0
    x += dx;
1439
0
    xi = x >> 16;
1440
0
    xf = (x >> 9) & 0x7f;
1441
0
    a = src[xi];
1442
0
    b = src[xi + 1];
1443
0
    dst[1] = BLENDER(a, b, xf);
1444
0
    x += dx;
1445
0
    dst += 2;
1446
0
  }
1447
0
  if (dst_width & 1) {
1448
0
    int64_t xi = x >> 16;
1449
0
    int xf = (x >> 9) & 0x7f;
1450
0
    uint16_t a = src[xi];
1451
0
    uint16_t b = src[xi + 1];
1452
0
    dst[0] = BLENDER(a, b, xf);
1453
0
  }
1454
0
}
1455
#undef BLENDER1
1456
#undef BLENDERC
1457
#undef BLENDER
1458
1459
// Scale plane vertically with bilinear interpolation.
1460
void ScalePlaneVertical(int src_height,
1461
                        int dst_width,
1462
                        int dst_height,
1463
                        int src_stride,
1464
                        int dst_stride,
1465
                        const uint8_t* src_argb,
1466
                        uint8_t* dst_argb,
1467
                        int x,
1468
                        int y,
1469
                        int dy,
1470
                        int bpp,  // bytes per pixel. 4 for ARGB.
1471
2.13k
                        enum FilterMode filtering) {
1472
  // TODO(fbarchard): Allow higher bpp.
1473
2.13k
  int dst_width_bytes = dst_width * bpp;
1474
2.13k
  void (*InterpolateRow)(uint8_t* dst_argb, const uint8_t* src_argb,
1475
2.13k
                         ptrdiff_t src_stride, int dst_width,
1476
2.13k
                         int source_y_fraction) = InterpolateRow_C;
1477
2.13k
  const int64_t max_y =
1478
2.13k
      (src_height > 1) ? (((int64_t)src_height - 1) << 16) - 1 : 0;
1479
2.13k
  int64_t y64 = y;
1480
2.13k
  int j;
1481
2.13k
  assert(bpp >= 1 && bpp <= 4);
1482
2.13k
  assert(src_height != 0);
1483
2.13k
  assert(dst_width > 0);
1484
2.13k
  assert(dst_height > 0);
1485
2.13k
  src_argb += (x >> 16) * bpp;
1486
2.13k
#if defined(HAS_INTERPOLATEROW_AVX2)
1487
2.13k
  if (TestCpuFlag(kCpuHasAVX2)) {
1488
2.13k
    InterpolateRow = InterpolateRow_Any_AVX2;
1489
2.13k
    if (IS_ALIGNED(dst_width_bytes, 32)) {
1490
518
      InterpolateRow = InterpolateRow_AVX2;
1491
518
    }
1492
2.13k
  }
1493
2.13k
#endif
1494
#if defined(HAS_INTERPOLATEROW_NEON)
1495
  if (TestCpuFlag(kCpuHasNEON)) {
1496
    InterpolateRow = InterpolateRow_Any_NEON;
1497
    if (IS_ALIGNED(dst_width_bytes, 16)) {
1498
      InterpolateRow = InterpolateRow_NEON;
1499
    }
1500
  }
1501
#endif
1502
#if defined(HAS_INTERPOLATEROW_SVE2)
1503
  if (TestCpuFlag(kCpuHasSVE2)) {
1504
    InterpolateRow = InterpolateRow_SVE2;
1505
  }
1506
#endif
1507
#if defined(HAS_INTERPOLATEROW_SME)
1508
  if (TestCpuFlag(kCpuHasSME)) {
1509
    InterpolateRow = InterpolateRow_SME;
1510
  }
1511
#endif
1512
#if defined(HAS_INTERPOLATEROW_LSX)
1513
  if (TestCpuFlag(kCpuHasLSX)) {
1514
    InterpolateRow = InterpolateRow_Any_LSX;
1515
    if (IS_ALIGNED(dst_width_bytes, 32)) {
1516
      InterpolateRow = InterpolateRow_LSX;
1517
    }
1518
  }
1519
#endif
1520
#if defined(HAS_INTERPOLATEROW_RVV)
1521
  if (TestCpuFlag(kCpuHasRVV)) {
1522
    InterpolateRow = InterpolateRow_RVV;
1523
  }
1524
#endif
1525
1526
1.52M
  for (j = 0; j < dst_height; ++j) {
1527
1.52M
    int yi;
1528
1.52M
    int yf;
1529
1.52M
    if (y64 > max_y) {
1530
0
      y64 = max_y;
1531
0
    }
1532
1.52M
    yi = (int)(y64 >> 16);
1533
1.52M
    yf = filtering ? (int)((y64 >> 8) & 255) : 0;
1534
1.52M
    InterpolateRow(dst_argb, src_argb + yi * (ptrdiff_t)src_stride, src_stride,
1535
1.52M
                   dst_width_bytes, yf);
1536
1.52M
    dst_argb += dst_stride;
1537
1.52M
    y64 += dy;
1538
1.52M
  }
1539
2.13k
}
1540
1541
void ScalePlaneVertical_16(int src_height,
1542
                           int dst_width,
1543
                           int dst_height,
1544
                           int src_stride,
1545
                           int dst_stride,
1546
                           const uint16_t* src_argb,
1547
                           uint16_t* dst_argb,
1548
                           int x,
1549
                           int y,
1550
                           int dy,
1551
                           int wpp, /* words per pixel. normally 1 */
1552
1.69k
                           enum FilterMode filtering) {
1553
  // TODO(fbarchard): Allow higher wpp.
1554
1.69k
  int dst_width_words = dst_width * wpp;
1555
1.69k
  void (*InterpolateRow)(uint16_t* dst_argb, const uint16_t* src_argb,
1556
1.69k
                         ptrdiff_t src_stride, int dst_width,
1557
1.69k
                         int source_y_fraction) = InterpolateRow_16_C;
1558
1.69k
  const int64_t max_y =
1559
1.69k
      (src_height > 1) ? (((int64_t)src_height - 1) << 16) - 1 : 0;
1560
1.69k
  int64_t y64 = y;
1561
1.69k
  int j;
1562
1.69k
  assert(wpp >= 1 && wpp <= 2);
1563
1.69k
  assert(src_height != 0);
1564
1.69k
  assert(dst_width > 0);
1565
1.69k
  assert(dst_height > 0);
1566
1.69k
  src_argb += (x >> 16) * wpp;
1567
#if defined(HAS_INTERPOLATEROW_16_SSSE3)
1568
  if (TestCpuFlag(kCpuHasSSSE3)) {
1569
    InterpolateRow = InterpolateRow_16_Any_SSSE3;
1570
    if (IS_ALIGNED(dst_width_words, 16)) {
1571
      InterpolateRow = InterpolateRow_16_SSSE3;
1572
    }
1573
  }
1574
#endif
1575
1.69k
#if defined(HAS_INTERPOLATEROW_16_AVX2)
1576
1.69k
  if (TestCpuFlag(kCpuHasAVX2)) {
1577
1.69k
    InterpolateRow = InterpolateRow_16_Any_AVX2;
1578
1.69k
    if (IS_ALIGNED(dst_width_words, 32)) {
1579
764
      InterpolateRow = InterpolateRow_16_AVX2;
1580
764
    }
1581
1.69k
  }
1582
1.69k
#endif
1583
#if defined(HAS_INTERPOLATEROW_16_NEON)
1584
  if (TestCpuFlag(kCpuHasNEON)) {
1585
    InterpolateRow = InterpolateRow_16_Any_NEON;
1586
    if (IS_ALIGNED(dst_width_words, 8)) {
1587
      InterpolateRow = InterpolateRow_16_NEON;
1588
    }
1589
  }
1590
#endif
1591
#if defined(HAS_INTERPOLATEROW_16_SME)
1592
  if (TestCpuFlag(kCpuHasSME)) {
1593
    InterpolateRow = InterpolateRow_16_SME;
1594
  }
1595
#endif
1596
1.16M
  for (j = 0; j < dst_height; ++j) {
1597
1.16M
    int yi;
1598
1.16M
    int yf;
1599
1.16M
    if (y64 > max_y) {
1600
0
      y64 = max_y;
1601
0
    }
1602
1.16M
    yi = (int)(y64 >> 16);
1603
1.16M
    yf = filtering ? (int)((y64 >> 8) & 255) : 0;
1604
1.16M
    InterpolateRow(dst_argb, src_argb + yi * (ptrdiff_t)src_stride, src_stride,
1605
1.16M
                   dst_width_words, yf);
1606
1.16M
    dst_argb += dst_stride;
1607
1.16M
    y64 += dy;
1608
1.16M
  }
1609
1.69k
}
1610
1611
// Simplify the filtering based on scale factors.
1612
enum FilterMode ScaleFilterReduce(int src_width,
1613
                                  int src_height,
1614
                                  int dst_width,
1615
                                  int dst_height,
1616
51.4k
                                  enum FilterMode filtering) {
1617
51.4k
  if (src_width < 0) {
1618
0
    src_width = -src_width;
1619
0
  }
1620
51.4k
  if (src_height < 0) {
1621
0
    src_height = -src_height;
1622
0
  }
1623
51.4k
  if (filtering == kFilterBox) {
1624
    // If scaling either axis to 0.5 or larger, switch from Box to Bilinear.
1625
35.4k
    if (dst_width * 2 >= src_width || dst_height * 2 >= src_height) {
1626
30.6k
      filtering = kFilterBilinear;
1627
30.6k
    }
1628
35.4k
  }
1629
51.4k
  if (filtering == kFilterBilinear) {
1630
42.9k
    if (src_height == 1) {
1631
3.83k
      filtering = kFilterLinear;
1632
3.83k
    }
1633
    // TODO(fbarchard): Detect any odd scale factor and reduce to Linear.
1634
42.9k
    if (dst_height == src_height || dst_height * 3 == src_height) {
1635
3.92k
      filtering = kFilterLinear;
1636
3.92k
    }
1637
    // TODO(fbarchard): Remove 1 pixel wide filter restriction, which is to
1638
    // avoid reading 2 pixels horizontally that causes memory exception.
1639
42.9k
    if (src_width == 1) {
1640
2.65k
      filtering = kFilterNone;
1641
2.65k
    }
1642
42.9k
  }
1643
51.4k
  if (filtering == kFilterLinear) {
1644
8.37k
    if (src_width == 1) {
1645
0
      filtering = kFilterNone;
1646
0
    }
1647
    // TODO(fbarchard): Detect any odd scale factor and reduce to None.
1648
8.37k
    if (dst_width == src_width || dst_width * 3 == src_width) {
1649
641
      filtering = kFilterNone;
1650
641
    }
1651
8.37k
  }
1652
51.4k
  return filtering;
1653
51.4k
}
1654
1655
// Divide num by div and return as 16.16 fixed point result.
1656
0
int FixedDiv_C(int num, int div) {
1657
0
  return (int)(((int64_t)(num) << 16) / div);
1658
0
}
1659
1660
// Divide num - 1 by div - 1 and return as 16.16 fixed point result.
1661
0
int FixedDiv1_C(int num, int div) {
1662
0
  return (int)((((int64_t)(num) << 16) - 0x00010001) / (div - 1));
1663
0
}
1664
1665
20.4k
#define CENTERSTART(dx, s) (dx < 0) ? -((-dx >> 1) + s) : ((dx >> 1) + s)
1666
1667
// Compute slope values for stepping.
1668
void ScaleSlope(int src_width,
1669
                int src_height,
1670
                int dst_width,
1671
                int dst_height,
1672
                enum FilterMode filtering,
1673
                int* x,
1674
                int* y,
1675
                int* dx,
1676
28.0k
                int* dy) {
1677
28.0k
  assert(x != NULL);
1678
28.0k
  assert(y != NULL);
1679
28.0k
  assert(dx != NULL);
1680
28.0k
  assert(dy != NULL);
1681
28.0k
  assert(src_width != 0);
1682
28.0k
  assert(src_height != 0);
1683
28.0k
  assert(dst_width > 0);
1684
28.0k
  assert(dst_height > 0);
1685
  // Check for 1 pixel and avoid FixedDiv overflow.
1686
28.0k
  if (dst_width == 1 && src_width >= 32768) {
1687
0
    dst_width = src_width;
1688
0
  }
1689
28.0k
  if (dst_height == 1 && src_height >= 32768) {
1690
0
    dst_height = src_height;
1691
0
  }
1692
28.0k
  if (filtering == kFilterBox) {
1693
    // Scale step for point sampling duplicates all pixels equally.
1694
2.90k
    *dx = FixedDiv(Abs(src_width), dst_width);
1695
2.90k
    *dy = FixedDiv(src_height, dst_height);
1696
2.90k
    *x = 0;
1697
2.90k
    *y = 0;
1698
25.1k
  } else if (filtering == kFilterBilinear) {
1699
    // Scale step for bilinear sampling renders last pixel once for upsample.
1700
17.7k
    if (dst_width <= Abs(src_width)) {
1701
4.55k
      *dx = FixedDiv(Abs(src_width), dst_width);
1702
4.55k
      *x = CENTERSTART(*dx, -32768);  // Subtract 0.5 (32768) to center filter.
1703
13.2k
    } else if (src_width > 1 && dst_width > 1) {
1704
13.2k
      *dx = FixedDiv1(Abs(src_width), dst_width);
1705
13.2k
      *x = 0;
1706
13.2k
    }
1707
17.7k
    if (dst_height <= src_height) {
1708
8.52k
      *dy = FixedDiv(src_height, dst_height);
1709
8.52k
      *y = CENTERSTART(*dy, -32768);  // Subtract 0.5 (32768) to center filter.
1710
9.22k
    } else if (src_height > 1 && dst_height > 1) {
1711
9.22k
      *dy = FixedDiv1(src_height, dst_height);
1712
9.22k
      *y = 0;
1713
9.22k
    }
1714
17.7k
  } else if (filtering == kFilterLinear) {
1715
    // Scale step for bilinear sampling renders last pixel once for upsample.
1716
4.99k
    if (dst_width <= Abs(src_width)) {
1717
2.66k
      *dx = FixedDiv(Abs(src_width), dst_width);
1718
2.66k
      *x = CENTERSTART(*dx, -32768);  // Subtract 0.5 (32768) to center filter.
1719
2.66k
    } else if (src_width > 1 && dst_width > 1) {
1720
2.32k
      *dx = FixedDiv1(Abs(src_width), dst_width);
1721
2.32k
      *x = 0;
1722
2.32k
    }
1723
4.99k
    *dy = FixedDiv(src_height, dst_height);
1724
4.99k
    *y = *dy >> 1;
1725
4.99k
  } else {
1726
    // Scale step for point sampling duplicates all pixels equally.
1727
2.35k
    *dx = FixedDiv(Abs(src_width), dst_width);
1728
2.35k
    *dy = FixedDiv(src_height, dst_height);
1729
2.35k
    *x = CENTERSTART(*dx, 0);
1730
2.35k
    *y = CENTERSTART(*dy, 0);
1731
2.35k
  }
1732
  // Negative src_width means horizontally mirror.
1733
28.0k
  if (src_width < 0) {
1734
0
    *x += (dst_width - 1) * *dx;
1735
0
    *dx = -*dx;
1736
    // src_width = -src_width;   // Caller must do this.
1737
0
  }
1738
28.0k
}
1739
#undef CENTERSTART
1740
1741
#ifdef __cplusplus
1742
}  // extern "C"
1743
}  // namespace libyuv
1744
#endif