Coverage Report

Created: 2026-09-03 06:27

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/libavif/ext/libyuv/source/scale.cc
Line
Count
Source
1
/*
2
 *  Copyright 2011 The LibYuv Project Authors. All rights reserved.
3
 *
4
 *  Use of this source code is governed by a BSD-style license
5
 *  that can be found in the LICENSE file in the root of the source
6
 *  tree. An additional intellectual property rights grant can be found
7
 *  in the file PATENTS. All contributing project authors may
8
 *  be found in the AUTHORS file in the root of the source tree.
9
 */
10
11
#include "libyuv/scale.h"
12
13
#include <assert.h>
14
#include <limits.h>
15
#include <string.h>
16
17
#include "libyuv/cpu_id.h"
18
#include "libyuv/planar_functions.h"  // For CopyPlane
19
#include "libyuv/row.h"
20
#include "libyuv/scale_row.h"
21
#include "libyuv/scale_uv.h"  // For UVScale
22
23
#ifdef __cplusplus
24
namespace libyuv {
25
extern "C" {
26
#endif
27
28
56.8k
static __inline int Abs(int v) {
29
56.8k
  return v >= 0 ? v : -v;
30
56.8k
}
31
32
0
#define SUBSAMPLE(v, a, s) (v < 0) ? (-((-v + a) >> s)) : ((v + a) >> s)
33
1.32k
#define CENTERSTART(dx, s) (dx < 0) ? -((-dx >> 1) + s) : ((dx >> 1) + s)
34
35
// Scale plane, 1/2
36
// This is an optimized version for scaling down a plane to 1/2 of
37
// its original size.
38
39
static void ScalePlaneDown2(int src_width,
40
                            int src_height,
41
                            int dst_width,
42
                            int dst_height,
43
                            ptrdiff_t src_stride,
44
                            ptrdiff_t dst_stride,
45
                            const uint8_t* src_ptr,
46
                            uint8_t* dst_ptr,
47
105
                            enum FilterMode filtering) {
48
105
  int y;
49
105
  void (*ScaleRowDown2)(const uint8_t* src_ptr, ptrdiff_t src_stride,
50
105
                        uint8_t* dst_ptr, int dst_width) =
51
105
      filtering == kFilterNone
52
105
          ? ScaleRowDown2_C
53
105
          : (filtering == kFilterLinear ? ScaleRowDown2Linear_C
54
105
                                        : ScaleRowDown2Box_C);
55
105
  ptrdiff_t row_stride = src_stride * 2;
56
105
  (void)src_width;
57
105
  (void)src_height;
58
105
  if (!filtering) {
59
0
    src_ptr += src_stride;  // Point to odd rows.
60
0
    src_stride = 0;
61
0
  }
62
63
#if defined(HAS_SCALEROWDOWN2_NEON)
64
  if (TestCpuFlag(kCpuHasNEON)) {
65
    ScaleRowDown2 =
66
        filtering == kFilterNone
67
            ? ScaleRowDown2_Any_NEON
68
            : (filtering == kFilterLinear ? ScaleRowDown2Linear_Any_NEON
69
                                          : ScaleRowDown2Box_Any_NEON);
70
    if (IS_ALIGNED(dst_width, 16)) {
71
      ScaleRowDown2 = filtering == kFilterNone ? ScaleRowDown2_NEON
72
                                               : (filtering == kFilterLinear
73
                                                      ? ScaleRowDown2Linear_NEON
74
                                                      : ScaleRowDown2Box_NEON);
75
    }
76
  }
77
#endif
78
#if defined(HAS_SCALEROWDOWN2_SME)
79
  if (TestCpuFlag(kCpuHasSME)) {
80
    ScaleRowDown2 = filtering == kFilterNone     ? ScaleRowDown2_SME
81
                    : filtering == kFilterLinear ? ScaleRowDown2Linear_SME
82
                                                 : ScaleRowDown2Box_SME;
83
  }
84
#endif
85
105
#if defined(HAS_SCALEROWDOWN2_SSSE3)
86
105
  if (TestCpuFlag(kCpuHasSSSE3)) {
87
105
    ScaleRowDown2 =
88
105
        filtering == kFilterNone
89
105
            ? ScaleRowDown2_Any_SSSE3
90
105
            : (filtering == kFilterLinear ? ScaleRowDown2Linear_Any_SSSE3
91
105
                                          : ScaleRowDown2Box_Any_SSSE3);
92
105
    if (IS_ALIGNED(dst_width, 16)) {
93
12
      ScaleRowDown2 =
94
12
          filtering == kFilterNone
95
12
              ? ScaleRowDown2_SSSE3
96
12
              : (filtering == kFilterLinear ? ScaleRowDown2Linear_SSSE3
97
12
                                            : ScaleRowDown2Box_SSSE3);
98
12
    }
99
105
  }
100
105
#endif
101
105
#if defined(HAS_SCALEROWDOWN2_AVX2)
102
105
  if (TestCpuFlag(kCpuHasAVX2)) {
103
105
    ScaleRowDown2 =
104
105
        filtering == kFilterNone
105
105
            ? ScaleRowDown2_Any_AVX2
106
105
            : (filtering == kFilterLinear ? ScaleRowDown2Linear_Any_AVX2
107
105
                                          : ScaleRowDown2Box_Any_AVX2);
108
105
    if (IS_ALIGNED(dst_width, 32)) {
109
12
      ScaleRowDown2 = filtering == kFilterNone ? ScaleRowDown2_AVX2
110
12
                                               : (filtering == kFilterLinear
111
12
                                                      ? ScaleRowDown2Linear_AVX2
112
12
                                                      : ScaleRowDown2Box_AVX2);
113
12
    }
114
105
  }
115
105
#endif
116
#if defined(HAS_SCALEROWDOWN2_LSX)
117
  if (TestCpuFlag(kCpuHasLSX)) {
118
    ScaleRowDown2 =
119
        filtering == kFilterNone
120
            ? ScaleRowDown2_Any_LSX
121
            : (filtering == kFilterLinear ? ScaleRowDown2Linear_Any_LSX
122
                                          : ScaleRowDown2Box_Any_LSX);
123
    if (IS_ALIGNED(dst_width, 32)) {
124
      ScaleRowDown2 = filtering == kFilterNone ? ScaleRowDown2_LSX
125
                                               : (filtering == kFilterLinear
126
                                                      ? ScaleRowDown2Linear_LSX
127
                                                      : ScaleRowDown2Box_LSX);
128
    }
129
  }
130
#endif
131
#if defined(HAS_SCALEROWDOWN2_RVV)
132
  if (TestCpuFlag(kCpuHasRVV)) {
133
    ScaleRowDown2 = filtering == kFilterNone
134
                        ? ScaleRowDown2_RVV
135
                        : (filtering == kFilterLinear ? ScaleRowDown2Linear_RVV
136
                                                      : ScaleRowDown2Box_RVV);
137
  }
138
#endif
139
140
105
  if (filtering == kFilterLinear) {
141
0
    src_stride = 0;
142
0
  }
143
  // TODO(fbarchard): Loop through source height to allow odd height.
144
960
  for (y = 0; y < dst_height; ++y) {
145
855
    ScaleRowDown2(src_ptr, src_stride, dst_ptr, dst_width);
146
855
    src_ptr += row_stride;
147
855
    dst_ptr += dst_stride;
148
855
  }
149
105
}
150
151
static void ScalePlaneDown2_16(int src_width,
152
                               int src_height,
153
                               int dst_width,
154
                               int dst_height,
155
                               ptrdiff_t src_stride,
156
                               ptrdiff_t dst_stride,
157
                               const uint16_t* src_ptr,
158
                               uint16_t* dst_ptr,
159
93
                               enum FilterMode filtering) {
160
93
  int y;
161
93
  void (*ScaleRowDown2)(const uint16_t* src_ptr, ptrdiff_t src_stride,
162
93
                        uint16_t* dst_ptr, int dst_width) =
163
93
      filtering == kFilterNone
164
93
          ? ScaleRowDown2_16_C
165
93
          : (filtering == kFilterLinear ? ScaleRowDown2Linear_16_C
166
93
                                        : ScaleRowDown2Box_16_C);
167
93
  ptrdiff_t row_stride = src_stride * 2;
168
93
  (void)src_width;
169
93
  (void)src_height;
170
93
  if (!filtering) {
171
0
    src_ptr += src_stride;  // Point to odd rows.
172
0
    src_stride = 0;
173
0
  }
174
175
#if defined(HAS_SCALEROWDOWN2_16_NEON)
176
  if (TestCpuFlag(kCpuHasNEON) && IS_ALIGNED(dst_width, 16)) {
177
    ScaleRowDown2 = filtering == kFilterNone     ? ScaleRowDown2_16_NEON
178
                    : filtering == kFilterLinear ? ScaleRowDown2Linear_16_NEON
179
                                                 : ScaleRowDown2Box_16_NEON;
180
  }
181
#endif
182
#if defined(HAS_SCALEROWDOWN2_16_SME)
183
  if (TestCpuFlag(kCpuHasSME)) {
184
    ScaleRowDown2 = filtering == kFilterNone     ? ScaleRowDown2_16_SME
185
                    : filtering == kFilterLinear ? ScaleRowDown2Linear_16_SME
186
                                                 : ScaleRowDown2Box_16_SME;
187
  }
188
#endif
189
#if defined(HAS_SCALEROWDOWN2_16_SSE2)
190
  if (TestCpuFlag(kCpuHasSSE2) && IS_ALIGNED(dst_width, 16)) {
191
    ScaleRowDown2 =
192
        filtering == kFilterNone
193
            ? ScaleRowDown2_16_SSE2
194
            : (filtering == kFilterLinear ? ScaleRowDown2Linear_16_SSE2
195
                                          : ScaleRowDown2Box_16_SSE2);
196
  }
197
#endif
198
199
93
  if (filtering == kFilterLinear) {
200
0
    src_stride = 0;
201
0
  }
202
  // TODO(fbarchard): Loop through source height to allow odd height.
203
837
  for (y = 0; y < dst_height; ++y) {
204
744
    ScaleRowDown2(src_ptr, src_stride, dst_ptr, dst_width);
205
744
    src_ptr += row_stride;
206
744
    dst_ptr += dst_stride;
207
744
  }
208
93
}
209
210
211
// Scale plane, 1/4
212
// This is an optimized version for scaling down a plane to 1/4 of
213
// its original size.
214
215
static void ScalePlaneDown4(int src_width,
216
                            int src_height,
217
                            int dst_width,
218
                            int dst_height,
219
                            ptrdiff_t src_stride,
220
                            ptrdiff_t dst_stride,
221
                            const uint8_t* src_ptr,
222
                            uint8_t* dst_ptr,
223
52
                            enum FilterMode filtering) {
224
52
  int y;
225
52
  void (*ScaleRowDown4)(const uint8_t* src_ptr, ptrdiff_t src_stride,
226
52
                        uint8_t* dst_ptr, int dst_width) =
227
52
      filtering ? ScaleRowDown4Box_C : ScaleRowDown4_C;
228
52
  ptrdiff_t row_stride = src_stride * 4;
229
52
  (void)src_width;
230
52
  (void)src_height;
231
52
  if (!filtering) {
232
0
    src_ptr += src_stride * 2;  // Point to row 2.
233
0
    src_stride = 0;
234
0
  }
235
#if defined(HAS_SCALEROWDOWN4_NEON)
236
  if (TestCpuFlag(kCpuHasNEON)) {
237
    ScaleRowDown4 =
238
        filtering ? ScaleRowDown4Box_Any_NEON : ScaleRowDown4_Any_NEON;
239
    if (IS_ALIGNED(dst_width, 16)) {
240
      ScaleRowDown4 = filtering ? ScaleRowDown4Box_NEON : ScaleRowDown4_NEON;
241
    }
242
  }
243
#endif
244
52
#if defined(HAS_SCALEROWDOWN4_SSSE3)
245
52
  if (TestCpuFlag(kCpuHasSSSE3)) {
246
52
    ScaleRowDown4 =
247
52
        filtering ? ScaleRowDown4Box_Any_SSSE3 : ScaleRowDown4_Any_SSSE3;
248
52
    if (IS_ALIGNED(dst_width, 8)) {
249
0
      ScaleRowDown4 = filtering ? ScaleRowDown4Box_SSSE3 : ScaleRowDown4_SSSE3;
250
0
    }
251
52
  }
252
52
#endif
253
52
#if defined(HAS_SCALEROWDOWN4_AVX2)
254
52
  if (TestCpuFlag(kCpuHasAVX2)) {
255
52
    ScaleRowDown4 =
256
52
        filtering ? ScaleRowDown4Box_Any_AVX2 : ScaleRowDown4_Any_AVX2;
257
52
    if (IS_ALIGNED(dst_width, 16)) {
258
0
      ScaleRowDown4 = filtering ? ScaleRowDown4Box_AVX2 : ScaleRowDown4_AVX2;
259
0
    }
260
52
  }
261
52
#endif
262
#if defined(HAS_SCALEROWDOWN4_LSX)
263
  if (TestCpuFlag(kCpuHasLSX)) {
264
    ScaleRowDown4 =
265
        filtering ? ScaleRowDown4Box_Any_LSX : ScaleRowDown4_Any_LSX;
266
    if (IS_ALIGNED(dst_width, 16)) {
267
      ScaleRowDown4 = filtering ? ScaleRowDown4Box_LSX : ScaleRowDown4_LSX;
268
    }
269
  }
270
#endif
271
#if defined(HAS_SCALEROWDOWN4_RVV)
272
  if (TestCpuFlag(kCpuHasRVV)) {
273
    ScaleRowDown4 = filtering ? ScaleRowDown4Box_RVV : ScaleRowDown4_RVV;
274
  }
275
#endif
276
277
52
  if (filtering == kFilterLinear) {
278
0
    src_stride = 0;
279
0
  }
280
547
  for (y = 0; y < dst_height; ++y) {
281
495
    ScaleRowDown4(src_ptr, src_stride, dst_ptr, dst_width);
282
495
    src_ptr += row_stride;
283
495
    dst_ptr += dst_stride;
284
495
  }
285
52
}
286
287
static void ScalePlaneDown4_16(int src_width,
288
                               int src_height,
289
                               int dst_width,
290
                               int dst_height,
291
                               ptrdiff_t src_stride,
292
                               ptrdiff_t dst_stride,
293
                               const uint16_t* src_ptr,
294
                               uint16_t* dst_ptr,
295
50
                               enum FilterMode filtering) {
296
50
  int y;
297
50
  void (*ScaleRowDown4)(const uint16_t* src_ptr, ptrdiff_t src_stride,
298
50
                        uint16_t* dst_ptr, int dst_width) =
299
50
      filtering ? ScaleRowDown4Box_16_C : ScaleRowDown4_16_C;
300
50
  ptrdiff_t row_stride = src_stride * 4;
301
50
  (void)src_width;
302
50
  (void)src_height;
303
50
  if (!filtering) {
304
0
    src_ptr += src_stride * 2;  // Point to row 2.
305
0
    src_stride = 0;
306
0
  }
307
#if defined(HAS_SCALEROWDOWN4_16_NEON)
308
  if (TestCpuFlag(kCpuHasNEON) && IS_ALIGNED(dst_width, 8)) {
309
    ScaleRowDown4 =
310
        filtering ? ScaleRowDown4Box_16_NEON : ScaleRowDown4_16_NEON;
311
  }
312
#endif
313
#if defined(HAS_SCALEROWDOWN4_16_SSE2)
314
  if (TestCpuFlag(kCpuHasSSE2) && IS_ALIGNED(dst_width, 8)) {
315
    ScaleRowDown4 =
316
        filtering ? ScaleRowDown4Box_16_SSE2 : ScaleRowDown4_16_SSE2;
317
  }
318
#endif
319
320
50
  if (filtering == kFilterLinear) {
321
0
    src_stride = 0;
322
0
  }
323
579
  for (y = 0; y < dst_height; ++y) {
324
529
    ScaleRowDown4(src_ptr, src_stride, dst_ptr, dst_width);
325
529
    src_ptr += row_stride;
326
529
    dst_ptr += dst_stride;
327
529
  }
328
50
}
329
330
// Scale plane down, 3/4
331
static void ScalePlaneDown34(int src_width,
332
                             int src_height,
333
                             int dst_width,
334
                             int dst_height,
335
                             ptrdiff_t src_stride,
336
                             ptrdiff_t dst_stride,
337
                             const uint8_t* src_ptr,
338
                             uint8_t* dst_ptr,
339
2
                             enum FilterMode filtering) {
340
2
  int y;
341
2
  void (*ScaleRowDown34_0)(const uint8_t* src_ptr, ptrdiff_t src_stride,
342
2
                           uint8_t* dst_ptr, int dst_width);
343
2
  void (*ScaleRowDown34_1)(const uint8_t* src_ptr, ptrdiff_t src_stride,
344
2
                           uint8_t* dst_ptr, int dst_width);
345
2
  const ptrdiff_t filter_stride = (filtering == kFilterLinear) ? 0 : src_stride;
346
2
  (void)src_width;
347
2
  (void)src_height;
348
2
  assert(dst_width % 3 == 0);
349
2
  if (!filtering) {
350
0
    ScaleRowDown34_0 = ScaleRowDown34_C;
351
0
    ScaleRowDown34_1 = ScaleRowDown34_C;
352
2
  } else {
353
2
    ScaleRowDown34_0 = ScaleRowDown34_0_Box_C;
354
2
    ScaleRowDown34_1 = ScaleRowDown34_1_Box_C;
355
2
  }
356
#if defined(HAS_SCALEROWDOWN34_NEON)
357
  if (TestCpuFlag(kCpuHasNEON)) {
358
#if defined(__aarch64__)
359
    if (dst_width % 48 == 0) {
360
#else
361
    if (dst_width % 24 == 0) {
362
#endif
363
      if (!filtering) {
364
        ScaleRowDown34_0 = ScaleRowDown34_NEON;
365
        ScaleRowDown34_1 = ScaleRowDown34_NEON;
366
      } else {
367
        ScaleRowDown34_0 = ScaleRowDown34_0_Box_NEON;
368
        ScaleRowDown34_1 = ScaleRowDown34_1_Box_NEON;
369
      }
370
    } else {
371
      if (!filtering) {
372
        ScaleRowDown34_0 = ScaleRowDown34_Any_NEON;
373
        ScaleRowDown34_1 = ScaleRowDown34_Any_NEON;
374
      } else {
375
        ScaleRowDown34_0 = ScaleRowDown34_0_Box_Any_NEON;
376
        ScaleRowDown34_1 = ScaleRowDown34_1_Box_Any_NEON;
377
      }
378
    }
379
  }
380
#endif
381
#if defined(HAS_SCALEROWDOWN34_LSX)
382
  if (TestCpuFlag(kCpuHasLSX)) {
383
    if (dst_width % 48 == 0) {
384
      if (!filtering) {
385
        ScaleRowDown34_0 = ScaleRowDown34_LSX;
386
        ScaleRowDown34_1 = ScaleRowDown34_LSX;
387
      } else {
388
        ScaleRowDown34_0 = ScaleRowDown34_0_Box_LSX;
389
        ScaleRowDown34_1 = ScaleRowDown34_1_Box_LSX;
390
      }
391
    } else {
392
      if (!filtering) {
393
        ScaleRowDown34_0 = ScaleRowDown34_Any_LSX;
394
        ScaleRowDown34_1 = ScaleRowDown34_Any_LSX;
395
      } else {
396
        ScaleRowDown34_0 = ScaleRowDown34_0_Box_Any_LSX;
397
        ScaleRowDown34_1 = ScaleRowDown34_1_Box_Any_LSX;
398
      }
399
    }
400
  }
401
#endif
402
2
#if defined(HAS_SCALEROWDOWN34_SSSE3)
403
2
  if (TestCpuFlag(kCpuHasSSSE3)) {
404
2
    if (dst_width % 24 == 0) {
405
0
      if (!filtering) {
406
0
        ScaleRowDown34_0 = ScaleRowDown34_SSSE3;
407
0
        ScaleRowDown34_1 = ScaleRowDown34_SSSE3;
408
0
      } else {
409
0
        ScaleRowDown34_0 = ScaleRowDown34_0_Box_SSSE3;
410
0
        ScaleRowDown34_1 = ScaleRowDown34_1_Box_SSSE3;
411
0
      }
412
2
    } else {
413
2
      if (!filtering) {
414
0
        ScaleRowDown34_0 = ScaleRowDown34_Any_SSSE3;
415
0
        ScaleRowDown34_1 = ScaleRowDown34_Any_SSSE3;
416
2
      } else {
417
2
        ScaleRowDown34_0 = ScaleRowDown34_0_Box_Any_SSSE3;
418
2
        ScaleRowDown34_1 = ScaleRowDown34_1_Box_Any_SSSE3;
419
2
      }
420
2
    }
421
2
  }
422
2
#endif
423
#if defined(HAS_SCALEROWDOWN34_RVV)
424
  if (TestCpuFlag(kCpuHasRVV)) {
425
    if (!filtering) {
426
      ScaleRowDown34_0 = ScaleRowDown34_RVV;
427
      ScaleRowDown34_1 = ScaleRowDown34_RVV;
428
    } else {
429
      ScaleRowDown34_0 = ScaleRowDown34_0_Box_RVV;
430
      ScaleRowDown34_1 = ScaleRowDown34_1_Box_RVV;
431
    }
432
  }
433
#endif
434
435
4
  for (y = 0; y < dst_height - 2; y += 3) {
436
2
    ScaleRowDown34_0(src_ptr, filter_stride, dst_ptr, dst_width);
437
2
    src_ptr += src_stride;
438
2
    dst_ptr += dst_stride;
439
2
    ScaleRowDown34_1(src_ptr, filter_stride, dst_ptr, dst_width);
440
2
    src_ptr += src_stride;
441
2
    dst_ptr += dst_stride;
442
2
    ScaleRowDown34_0(src_ptr + src_stride, -filter_stride, dst_ptr, dst_width);
443
2
    src_ptr += src_stride * 2;
444
2
    dst_ptr += dst_stride;
445
2
  }
446
447
  // Remainder 1 or 2 rows with last row vertically unfiltered
448
2
  if ((dst_height % 3) == 2) {
449
0
    ScaleRowDown34_0(src_ptr, filter_stride, dst_ptr, dst_width);
450
0
    src_ptr += src_stride;
451
0
    dst_ptr += dst_stride;
452
0
    ScaleRowDown34_1(src_ptr, 0, dst_ptr, dst_width);
453
2
  } else if ((dst_height % 3) == 1) {
454
0
    ScaleRowDown34_0(src_ptr, 0, dst_ptr, dst_width);
455
0
  }
456
2
}
457
458
static void ScalePlaneDown34_16(int src_width,
459
                                int src_height,
460
                                int dst_width,
461
                                int dst_height,
462
                                ptrdiff_t src_stride,
463
                                ptrdiff_t dst_stride,
464
                                const uint16_t* src_ptr,
465
                                uint16_t* dst_ptr,
466
4
                                enum FilterMode filtering) {
467
4
  int y;
468
4
  void (*ScaleRowDown34_0)(const uint16_t* src_ptr, ptrdiff_t src_stride,
469
4
                           uint16_t* dst_ptr, int dst_width);
470
4
  void (*ScaleRowDown34_1)(const uint16_t* src_ptr, ptrdiff_t src_stride,
471
4
                           uint16_t* dst_ptr, int dst_width);
472
4
  const ptrdiff_t filter_stride = (filtering == kFilterLinear) ? 0 : src_stride;
473
4
  (void)src_width;
474
4
  (void)src_height;
475
4
  assert(dst_width % 3 == 0);
476
4
  if (!filtering) {
477
0
    ScaleRowDown34_0 = ScaleRowDown34_16_C;
478
0
    ScaleRowDown34_1 = ScaleRowDown34_16_C;
479
4
  } else {
480
4
    ScaleRowDown34_0 = ScaleRowDown34_0_Box_16_C;
481
4
    ScaleRowDown34_1 = ScaleRowDown34_1_Box_16_C;
482
4
  }
483
#if defined(HAS_SCALEROWDOWN34_16_NEON)
484
  if (TestCpuFlag(kCpuHasNEON) && (dst_width % 24 == 0)) {
485
    if (!filtering) {
486
      ScaleRowDown34_0 = ScaleRowDown34_16_NEON;
487
      ScaleRowDown34_1 = ScaleRowDown34_16_NEON;
488
    } else {
489
      ScaleRowDown34_0 = ScaleRowDown34_0_Box_16_NEON;
490
      ScaleRowDown34_1 = ScaleRowDown34_1_Box_16_NEON;
491
    }
492
  }
493
#endif
494
#if defined(HAS_SCALEROWDOWN34_16_SSSE3)
495
  if (TestCpuFlag(kCpuHasSSSE3) && (dst_width % 24 == 0)) {
496
    if (!filtering) {
497
      ScaleRowDown34_0 = ScaleRowDown34_16_SSSE3;
498
      ScaleRowDown34_1 = ScaleRowDown34_16_SSSE3;
499
    } else {
500
      ScaleRowDown34_0 = ScaleRowDown34_0_Box_16_SSSE3;
501
      ScaleRowDown34_1 = ScaleRowDown34_1_Box_16_SSSE3;
502
    }
503
  }
504
#endif
505
506
28
  for (y = 0; y < dst_height - 2; y += 3) {
507
24
    ScaleRowDown34_0(src_ptr, filter_stride, dst_ptr, dst_width);
508
24
    src_ptr += src_stride;
509
24
    dst_ptr += dst_stride;
510
24
    ScaleRowDown34_1(src_ptr, filter_stride, dst_ptr, dst_width);
511
24
    src_ptr += src_stride;
512
24
    dst_ptr += dst_stride;
513
24
    ScaleRowDown34_0(src_ptr + src_stride, -filter_stride, dst_ptr, dst_width);
514
24
    src_ptr += src_stride * 2;
515
24
    dst_ptr += dst_stride;
516
24
  }
517
518
  // Remainder 1 or 2 rows with last row vertically unfiltered
519
4
  if ((dst_height % 3) == 2) {
520
0
    ScaleRowDown34_0(src_ptr, filter_stride, dst_ptr, dst_width);
521
0
    src_ptr += src_stride;
522
0
    dst_ptr += dst_stride;
523
0
    ScaleRowDown34_1(src_ptr, 0, dst_ptr, dst_width);
524
4
  } else if ((dst_height % 3) == 1) {
525
0
    ScaleRowDown34_0(src_ptr, 0, dst_ptr, dst_width);
526
0
  }
527
4
}
528
529
// Scale plane, 3/8
530
// This is an optimized version for scaling down a plane to 3/8
531
// of its original size.
532
//
533
// Uses box filter arranges like this
534
// aaabbbcc -> abc
535
// aaabbbcc    def
536
// aaabbbcc    ghi
537
// dddeeeff
538
// dddeeeff
539
// dddeeeff
540
// ggghhhii
541
// ggghhhii
542
// Boxes are 3x3, 2x3, 3x2 and 2x2
543
544
static void ScalePlaneDown38(int src_width,
545
                             int src_height,
546
                             int dst_width,
547
                             int dst_height,
548
                             ptrdiff_t src_stride,
549
                             ptrdiff_t dst_stride,
550
                             const uint8_t* src_ptr,
551
                             uint8_t* dst_ptr,
552
12
                             enum FilterMode filtering) {
553
12
  int y;
554
12
  void (*ScaleRowDown38_3)(const uint8_t* src_ptr, ptrdiff_t src_stride,
555
12
                           uint8_t* dst_ptr, int dst_width);
556
12
  void (*ScaleRowDown38_2)(const uint8_t* src_ptr, ptrdiff_t src_stride,
557
12
                           uint8_t* dst_ptr, int dst_width);
558
12
  const ptrdiff_t filter_stride = (filtering == kFilterLinear) ? 0 : src_stride;
559
12
  assert(dst_width % 3 == 0);
560
12
  (void)src_width;
561
12
  (void)src_height;
562
12
  if (!filtering) {
563
0
    ScaleRowDown38_3 = ScaleRowDown38_C;
564
0
    ScaleRowDown38_2 = ScaleRowDown38_C;
565
12
  } else {
566
12
    ScaleRowDown38_3 = ScaleRowDown38_3_Box_C;
567
12
    ScaleRowDown38_2 = ScaleRowDown38_2_Box_C;
568
12
  }
569
570
#if defined(HAS_SCALEROWDOWN38_NEON)
571
  if (TestCpuFlag(kCpuHasNEON)) {
572
    if (!filtering) {
573
      ScaleRowDown38_3 = ScaleRowDown38_Any_NEON;
574
      ScaleRowDown38_2 = ScaleRowDown38_Any_NEON;
575
    } else {
576
      ScaleRowDown38_3 = ScaleRowDown38_3_Box_Any_NEON;
577
      ScaleRowDown38_2 = ScaleRowDown38_2_Box_Any_NEON;
578
    }
579
    if (dst_width % 12 == 0) {
580
      if (!filtering) {
581
        ScaleRowDown38_3 = ScaleRowDown38_NEON;
582
        ScaleRowDown38_2 = ScaleRowDown38_NEON;
583
      } else {
584
        ScaleRowDown38_3 = ScaleRowDown38_3_Box_NEON;
585
        ScaleRowDown38_2 = ScaleRowDown38_2_Box_NEON;
586
      }
587
    }
588
  }
589
#endif
590
12
#if defined(HAS_SCALEROWDOWN38_SSSE3)
591
12
  if (TestCpuFlag(kCpuHasSSSE3)) {
592
12
    if (!filtering) {
593
0
      ScaleRowDown38_3 = ScaleRowDown38_Any_SSSE3;
594
0
      ScaleRowDown38_2 = ScaleRowDown38_Any_SSSE3;
595
12
    } else {
596
12
      ScaleRowDown38_3 = ScaleRowDown38_3_Box_Any_SSSE3;
597
12
      ScaleRowDown38_2 = ScaleRowDown38_2_Box_Any_SSSE3;
598
12
    }
599
12
    if (dst_width % 12 == 0 && !filtering) {
600
0
      ScaleRowDown38_3 = ScaleRowDown38_SSSE3;
601
0
      ScaleRowDown38_2 = ScaleRowDown38_SSSE3;
602
0
    }
603
12
    if (dst_width % 6 == 0 && filtering) {
604
12
      ScaleRowDown38_3 = ScaleRowDown38_3_Box_SSSE3;
605
12
      ScaleRowDown38_2 = ScaleRowDown38_2_Box_SSSE3;
606
12
    }
607
12
  }
608
12
#endif
609
#if defined(HAS_SCALEROWDOWN38_LSX)
610
  if (TestCpuFlag(kCpuHasLSX)) {
611
    if (!filtering) {
612
      ScaleRowDown38_3 = ScaleRowDown38_Any_LSX;
613
      ScaleRowDown38_2 = ScaleRowDown38_Any_LSX;
614
    } else {
615
      ScaleRowDown38_3 = ScaleRowDown38_3_Box_Any_LSX;
616
      ScaleRowDown38_2 = ScaleRowDown38_2_Box_Any_LSX;
617
    }
618
    if (dst_width % 12 == 0) {
619
      if (!filtering) {
620
        ScaleRowDown38_3 = ScaleRowDown38_LSX;
621
        ScaleRowDown38_2 = ScaleRowDown38_LSX;
622
      } else {
623
        ScaleRowDown38_3 = ScaleRowDown38_3_Box_LSX;
624
        ScaleRowDown38_2 = ScaleRowDown38_2_Box_LSX;
625
      }
626
    }
627
  }
628
#endif
629
#if defined(HAS_SCALEROWDOWN38_RVV)
630
  if (TestCpuFlag(kCpuHasRVV)) {
631
    if (!filtering) {
632
      ScaleRowDown38_3 = ScaleRowDown38_RVV;
633
      ScaleRowDown38_2 = ScaleRowDown38_RVV;
634
    } else {
635
      ScaleRowDown38_3 = ScaleRowDown38_3_Box_RVV;
636
      ScaleRowDown38_2 = ScaleRowDown38_2_Box_RVV;
637
    }
638
  }
639
#endif
640
641
36
  for (y = 0; y < dst_height - 2; y += 3) {
642
24
    ScaleRowDown38_3(src_ptr, filter_stride, dst_ptr, dst_width);
643
24
    src_ptr += src_stride * 3;
644
24
    dst_ptr += dst_stride;
645
24
    ScaleRowDown38_3(src_ptr, filter_stride, dst_ptr, dst_width);
646
24
    src_ptr += src_stride * 3;
647
24
    dst_ptr += dst_stride;
648
24
    ScaleRowDown38_2(src_ptr, filter_stride, dst_ptr, dst_width);
649
24
    src_ptr += src_stride * 2;
650
24
    dst_ptr += dst_stride;
651
24
  }
652
653
  // Remainder 1 or 2 rows with last row vertically unfiltered
654
12
  if ((dst_height % 3) == 2) {
655
0
    ScaleRowDown38_3(src_ptr, filter_stride, dst_ptr, dst_width);
656
0
    src_ptr += src_stride * 3;
657
0
    dst_ptr += dst_stride;
658
0
    ScaleRowDown38_3(src_ptr, 0, dst_ptr, dst_width);
659
12
  } else if ((dst_height % 3) == 1) {
660
0
    ScaleRowDown38_3(src_ptr, 0, dst_ptr, dst_width);
661
0
  }
662
12
}
663
664
static void ScalePlaneDown38_16(int src_width,
665
                                int src_height,
666
                                int dst_width,
667
                                int dst_height,
668
                                ptrdiff_t src_stride,
669
                                ptrdiff_t dst_stride,
670
                                const uint16_t* src_ptr,
671
                                uint16_t* dst_ptr,
672
12
                                enum FilterMode filtering) {
673
12
  int y;
674
12
  void (*ScaleRowDown38_3)(const uint16_t* src_ptr, ptrdiff_t src_stride,
675
12
                           uint16_t* dst_ptr, int dst_width);
676
12
  void (*ScaleRowDown38_2)(const uint16_t* src_ptr, ptrdiff_t src_stride,
677
12
                           uint16_t* dst_ptr, int dst_width);
678
12
  const ptrdiff_t filter_stride = (filtering == kFilterLinear) ? 0 : src_stride;
679
12
  (void)src_width;
680
12
  (void)src_height;
681
12
  assert(dst_width % 3 == 0);
682
12
  if (!filtering) {
683
0
    ScaleRowDown38_3 = ScaleRowDown38_16_C;
684
0
    ScaleRowDown38_2 = ScaleRowDown38_16_C;
685
12
  } else {
686
12
    ScaleRowDown38_3 = ScaleRowDown38_3_Box_16_C;
687
12
    ScaleRowDown38_2 = ScaleRowDown38_2_Box_16_C;
688
12
  }
689
#if defined(HAS_SCALEROWDOWN38_16_NEON)
690
  if (TestCpuFlag(kCpuHasNEON) && (dst_width % 12 == 0)) {
691
    if (!filtering) {
692
      ScaleRowDown38_3 = ScaleRowDown38_16_NEON;
693
      ScaleRowDown38_2 = ScaleRowDown38_16_NEON;
694
    } else {
695
      ScaleRowDown38_3 = ScaleRowDown38_3_Box_16_NEON;
696
      ScaleRowDown38_2 = ScaleRowDown38_2_Box_16_NEON;
697
    }
698
  }
699
#endif
700
#if defined(HAS_SCALEROWDOWN38_16_SSSE3)
701
  if (TestCpuFlag(kCpuHasSSSE3) && (dst_width % 24 == 0)) {
702
    if (!filtering) {
703
      ScaleRowDown38_3 = ScaleRowDown38_16_SSSE3;
704
      ScaleRowDown38_2 = ScaleRowDown38_16_SSSE3;
705
    } else {
706
      ScaleRowDown38_3 = ScaleRowDown38_3_Box_16_SSSE3;
707
      ScaleRowDown38_2 = ScaleRowDown38_2_Box_16_SSSE3;
708
    }
709
  }
710
#endif
711
712
36
  for (y = 0; y < dst_height - 2; y += 3) {
713
24
    ScaleRowDown38_3(src_ptr, filter_stride, dst_ptr, dst_width);
714
24
    src_ptr += src_stride * 3;
715
24
    dst_ptr += dst_stride;
716
24
    ScaleRowDown38_3(src_ptr, filter_stride, dst_ptr, dst_width);
717
24
    src_ptr += src_stride * 3;
718
24
    dst_ptr += dst_stride;
719
24
    ScaleRowDown38_2(src_ptr, filter_stride, dst_ptr, dst_width);
720
24
    src_ptr += src_stride * 2;
721
24
    dst_ptr += dst_stride;
722
24
  }
723
724
  // Remainder 1 or 2 rows with last row vertically unfiltered
725
12
  if ((dst_height % 3) == 2) {
726
0
    ScaleRowDown38_3(src_ptr, filter_stride, dst_ptr, dst_width);
727
0
    src_ptr += src_stride * 3;
728
0
    dst_ptr += dst_stride;
729
0
    ScaleRowDown38_3(src_ptr, 0, dst_ptr, dst_width);
730
12
  } else if ((dst_height % 3) == 1) {
731
0
    ScaleRowDown38_3(src_ptr, 0, dst_ptr, dst_width);
732
0
  }
733
12
}
734
735
6.72M
#define MIN1(x) ((x) < 1 ? 1 : (x))
736
737
4.57M
static __inline uint32_t SumPixels(int iboxwidth, const uint16_t* src_ptr) {
738
4.57M
  uint32_t sum = 0u;
739
4.57M
  int x;
740
4.57M
  assert(iboxwidth > 0);
741
26.7M
  for (x = 0; x < iboxwidth; ++x) {
742
22.1M
    sum += src_ptr[x];
743
22.1M
  }
744
4.57M
  return sum;
745
4.57M
}
746
747
5.16M
static __inline uint32_t SumPixels_16(int iboxwidth, const uint32_t* src_ptr) {
748
5.16M
  uint32_t sum = 0u;
749
5.16M
  int x;
750
5.16M
  assert(iboxwidth > 0);
751
33.3M
  for (x = 0; x < iboxwidth; ++x) {
752
28.1M
    sum += src_ptr[x];
753
28.1M
  }
754
5.16M
  return sum;
755
5.16M
}
756
757
static void ScaleAddCols2_C(int dst_width,
758
                            int boxheight,
759
                            int x,
760
                            int dx,
761
                            const uint16_t* src_ptr,
762
39.6k
                            uint8_t* dst_ptr) {
763
39.6k
  int i;
764
39.6k
  int scaletbl[2];
765
39.6k
  int minboxwidth = dx >> 16;
766
39.6k
  int boxwidth;
767
39.6k
  scaletbl[0] = 65536 / (MIN1(minboxwidth) * boxheight);
768
39.6k
  scaletbl[1] = 65536 / (MIN1(minboxwidth + 1) * boxheight);
769
2.84M
  for (i = 0; i < dst_width; ++i) {
770
2.80M
    int ix = x >> 16;
771
2.80M
    x += dx;
772
2.80M
    boxwidth = MIN1((x >> 16) - ix);
773
2.80M
    int scaletbl_index = boxwidth - minboxwidth;
774
2.80M
    assert((scaletbl_index == 0) || (scaletbl_index == 1));
775
2.80M
    *dst_ptr++ = (uint8_t)(SumPixels(boxwidth, src_ptr + ix) *
776
2.80M
                               scaletbl[scaletbl_index] >>
777
2.80M
                           16);
778
2.80M
  }
779
39.6k
}
780
781
static void ScaleAddCols2_16_C(int dst_width,
782
                               int boxheight,
783
                               int x,
784
                               int dx,
785
                               const uint32_t* src_ptr,
786
49.9k
                               uint16_t* dst_ptr) {
787
49.9k
  int i;
788
49.9k
  int scaletbl[2];
789
49.9k
  int minboxwidth = dx >> 16;
790
49.9k
  int boxwidth;
791
49.9k
  scaletbl[0] = 65536 / (MIN1(minboxwidth) * boxheight);
792
49.9k
  scaletbl[1] = 65536 / (MIN1(minboxwidth + 1) * boxheight);
793
3.61M
  for (i = 0; i < dst_width; ++i) {
794
3.56M
    int ix = x >> 16;
795
3.56M
    x += dx;
796
3.56M
    boxwidth = MIN1((x >> 16) - ix);
797
3.56M
    int scaletbl_index = boxwidth - minboxwidth;
798
3.56M
    assert((scaletbl_index == 0) || (scaletbl_index == 1));
799
3.56M
    *dst_ptr++ =
800
3.56M
        SumPixels_16(boxwidth, src_ptr + ix) * scaletbl[scaletbl_index] >> 16;
801
3.56M
  }
802
49.9k
}
803
804
static void ScaleAddCols0_C(int dst_width,
805
                            int boxheight,
806
                            int x,
807
                            int dx,
808
                            const uint16_t* src_ptr,
809
0
                            uint8_t* dst_ptr) {
810
0
  int scaleval = 65536 / boxheight;
811
0
  int i;
812
0
  (void)dx;
813
0
  src_ptr += (x >> 16);
814
0
  for (i = 0; i < dst_width; ++i) {
815
0
    *dst_ptr++ = (uint8_t)(src_ptr[i] * scaleval >> 16);
816
0
  }
817
0
}
818
819
static void ScaleAddCols1_C(int dst_width,
820
                            int boxheight,
821
                            int x,
822
                            int dx,
823
                            const uint16_t* src_ptr,
824
17.9k
                            uint8_t* dst_ptr) {
825
17.9k
  int boxwidth = MIN1(dx >> 16);
826
17.9k
  int scaleval = 65536 / (boxwidth * boxheight);
827
17.9k
  int i;
828
17.9k
  x >>= 16;
829
1.79M
  for (i = 0; i < dst_width; ++i) {
830
1.77M
    *dst_ptr++ = (uint8_t)(SumPixels(boxwidth, src_ptr + x) * scaleval >> 16);
831
1.77M
    x += boxwidth;
832
1.77M
  }
833
17.9k
}
834
835
static void ScaleAddCols1_16_C(int dst_width,
836
                               int boxheight,
837
                               int x,
838
                               int dx,
839
                               const uint32_t* src_ptr,
840
30.2k
                               uint16_t* dst_ptr) {
841
30.2k
  int boxwidth = MIN1(dx >> 16);
842
30.2k
  int scaleval = 65536 / (boxwidth * boxheight);
843
30.2k
  int i;
844
1.62M
  for (i = 0; i < dst_width; ++i) {
845
1.59M
    *dst_ptr++ = SumPixels_16(boxwidth, src_ptr + x) * scaleval >> 16;
846
1.59M
    x += boxwidth;
847
1.59M
  }
848
30.2k
}
849
850
// Scale plane down to any dimensions, with interpolation.
851
// (boxfilter).
852
//
853
// Same method as SimpleScale, which is fixed point, outputting
854
// one pixel of destination using fixed point (16.16) to step
855
// through source, sampling a box of pixel with simple
856
// averaging.
857
static int ScalePlaneBox(int src_width,
858
                         int src_height,
859
                         int dst_width,
860
                         int dst_height,
861
                         ptrdiff_t src_stride,
862
                         ptrdiff_t dst_stride,
863
                         const uint8_t* src_ptr,
864
1.14k
                         uint8_t* dst_ptr) {
865
1.14k
  int j, k;
866
  // Initial source x/y coordinate and step values as 16.16 fixed point.
867
1.14k
  int x = 0;
868
1.14k
  int y = 0;
869
1.14k
  int dx = 0;
870
1.14k
  int dy = 0;
871
1.14k
  const int max_y = (src_height << 16);
872
1.14k
  ScaleSlope(src_width, src_height, dst_width, dst_height, kFilterBox, &x, &y,
873
1.14k
             &dx, &dy);
874
1.14k
  src_width = Abs(src_width);
875
1.14k
  {
876
    // Allocate a row buffer of uint16_t.
877
1.14k
    align_buffer_64(row16, src_width * 2);
878
1.14k
    if (!row16)
879
0
      return 1;
880
1.14k
    void (*ScaleAddCols)(int dst_width, int boxheight, int x, int dx,
881
1.14k
                         const uint16_t* src_ptr, uint8_t* dst_ptr) =
882
1.14k
        (dx & 0xffff) ? ScaleAddCols2_C
883
1.14k
                      : ((dx != 0x10000) ? ScaleAddCols1_C : ScaleAddCols0_C);
884
1.14k
    void (*ScaleAddRow)(const uint8_t* src_ptr, uint16_t* dst_ptr,
885
1.14k
                        int src_width) = ScaleAddRow_C;
886
1.14k
#if defined(HAS_SCALEADDROW_SSE2)
887
1.14k
    if (TestCpuFlag(kCpuHasSSE2)) {
888
1.14k
      ScaleAddRow = ScaleAddRow_Any_SSE2;
889
1.14k
      if (IS_ALIGNED(src_width, 16)) {
890
156
        ScaleAddRow = ScaleAddRow_SSE2;
891
156
      }
892
1.14k
    }
893
1.14k
#endif
894
1.14k
#if defined(HAS_SCALEADDROW_AVX2)
895
1.14k
    if (TestCpuFlag(kCpuHasAVX2)) {
896
1.14k
      ScaleAddRow = ScaleAddRow_Any_AVX2;
897
1.14k
      if (IS_ALIGNED(src_width, 32)) {
898
119
        ScaleAddRow = ScaleAddRow_AVX2;
899
119
      }
900
1.14k
    }
901
1.14k
#endif
902
#if defined(HAS_SCALEADDROW_NEON)
903
    if (TestCpuFlag(kCpuHasNEON)) {
904
      ScaleAddRow = ScaleAddRow_Any_NEON;
905
      if (IS_ALIGNED(src_width, 16)) {
906
        ScaleAddRow = ScaleAddRow_NEON;
907
      }
908
    }
909
#endif
910
#if defined(HAS_SCALEADDROW_LSX)
911
    if (TestCpuFlag(kCpuHasLSX)) {
912
      ScaleAddRow = ScaleAddRow_Any_LSX;
913
      if (IS_ALIGNED(src_width, 16)) {
914
        ScaleAddRow = ScaleAddRow_LSX;
915
      }
916
    }
917
#endif
918
#if defined(HAS_SCALEADDROW_RVV)
919
    if (TestCpuFlag(kCpuHasRVV)) {
920
      ScaleAddRow = ScaleAddRow_RVV;
921
    }
922
#endif
923
924
58.7k
    for (j = 0; j < dst_height; ++j) {
925
57.6k
      int boxheight;
926
57.6k
      int iy = y >> 16;
927
57.6k
      const uint8_t* src = src_ptr + iy * src_stride;
928
57.6k
      y += dy;
929
57.6k
      if (y > max_y) {
930
0
        y = max_y;
931
0
      }
932
57.6k
      boxheight = MIN1((y >> 16) - iy);
933
57.6k
      memset(row16, 0, src_width * 2);
934
705k
      for (k = 0; k < boxheight; ++k) {
935
647k
        ScaleAddRow(src, (uint16_t*)(row16), src_width);
936
647k
        src += src_stride;
937
647k
      }
938
57.6k
      ScaleAddCols(dst_width, boxheight, x, dx, (uint16_t*)(row16), dst_ptr);
939
57.6k
      dst_ptr += dst_stride;
940
57.6k
    }
941
1.14k
    free_aligned_buffer_64(row16);
942
1.14k
  }
943
0
  return 0;
944
1.14k
}
945
946
static int ScalePlaneBox_16(int src_width,
947
                            int src_height,
948
                            int dst_width,
949
                            int dst_height,
950
                            ptrdiff_t src_stride,
951
                            ptrdiff_t dst_stride,
952
                            const uint16_t* src_ptr,
953
1.75k
                            uint16_t* dst_ptr) {
954
1.75k
  int j, k;
955
  // Initial source x/y coordinate and step values as 16.16 fixed point.
956
1.75k
  int x = 0;
957
1.75k
  int y = 0;
958
1.75k
  int dx = 0;
959
1.75k
  int dy = 0;
960
1.75k
  const int max_y = (src_height << 16);
961
1.75k
  ScaleSlope(src_width, src_height, dst_width, dst_height, kFilterBox, &x, &y,
962
1.75k
             &dx, &dy);
963
1.75k
  src_width = Abs(src_width);
964
1.75k
  {
965
    // Allocate a row buffer of uint32_t.
966
1.75k
    align_buffer_64(row32, src_width * 4);
967
1.75k
    if (!row32)
968
0
      return 1;
969
1.75k
    void (*ScaleAddCols)(int dst_width, int boxheight, int x, int dx,
970
1.75k
                         const uint32_t* src_ptr, uint16_t* dst_ptr) =
971
1.75k
        (dx & 0xffff) ? ScaleAddCols2_16_C : ScaleAddCols1_16_C;
972
1.75k
    void (*ScaleAddRow)(const uint16_t* src_ptr, uint32_t* dst_ptr,
973
1.75k
                        int src_width) = ScaleAddRow_16_C;
974
975
#if defined(HAS_SCALEADDROW_16_SSE2)
976
    if (TestCpuFlag(kCpuHasSSE2) && IS_ALIGNED(src_width, 16)) {
977
      ScaleAddRow = ScaleAddRow_16_SSE2;
978
    }
979
#endif
980
981
81.9k
    for (j = 0; j < dst_height; ++j) {
982
80.2k
      int boxheight;
983
80.2k
      int iy = y >> 16;
984
80.2k
      const uint16_t* src = src_ptr + iy * src_stride;
985
80.2k
      y += dy;
986
80.2k
      if (y > max_y) {
987
0
        y = max_y;
988
0
      }
989
80.2k
      boxheight = MIN1((y >> 16) - iy);
990
80.2k
      memset(row32, 0, src_width * 4);
991
1.31M
      for (k = 0; k < boxheight; ++k) {
992
1.23M
        ScaleAddRow(src, (uint32_t*)(row32), src_width);
993
1.23M
        src += src_stride;
994
1.23M
      }
995
80.2k
      ScaleAddCols(dst_width, boxheight, x, dx, (uint32_t*)(row32), dst_ptr);
996
80.2k
      dst_ptr += dst_stride;
997
80.2k
    }
998
1.75k
    free_aligned_buffer_64(row32);
999
1.75k
  }
1000
0
  return 0;
1001
1.75k
}
1002
1003
// Scale plane down with bilinear interpolation.
1004
static int ScalePlaneBilinearDown(int src_width,
1005
                                  int src_height,
1006
                                  int dst_width,
1007
                                  int dst_height,
1008
                                  ptrdiff_t src_stride,
1009
                                  ptrdiff_t dst_stride,
1010
                                  const uint8_t* src_ptr,
1011
                                  uint8_t* dst_ptr,
1012
4.48k
                                  enum FilterMode filtering) {
1013
  // Initial source x/y coordinate and step values as 16.16 fixed point.
1014
4.48k
  int x = 0;
1015
4.48k
  int y = 0;
1016
4.48k
  int dx = 0;
1017
4.48k
  int dy = 0;
1018
  // TODO(fbarchard): Consider not allocating row buffer for kFilterLinear.
1019
  // Allocate a row buffer.
1020
4.48k
  align_buffer_64(row, src_width);
1021
4.48k
  if (!row)
1022
0
    return 1;
1023
1024
4.48k
  const int max_y = (src_height - 1) << 16;
1025
4.48k
  int j;
1026
4.48k
  void (*ScaleFilterCols)(uint8_t* dst_ptr, const uint8_t* src_ptr,
1027
4.48k
                          int dst_width, int x, int dx) =
1028
4.48k
      (src_width >= 32768) ? ScaleFilterCols64_C : ScaleFilterCols_C;
1029
4.48k
  void (*InterpolateRow)(uint8_t* dst_ptr, const uint8_t* src_ptr,
1030
4.48k
                         ptrdiff_t src_stride, int dst_width,
1031
4.48k
                         int source_y_fraction) = InterpolateRow_C;
1032
4.48k
  ScaleSlope(src_width, src_height, dst_width, dst_height, filtering, &x, &y,
1033
4.48k
             &dx, &dy);
1034
4.48k
  src_width = Abs(src_width);
1035
1036
4.48k
#if defined(HAS_INTERPOLATEROW_AVX2)
1037
4.48k
  if (TestCpuFlag(kCpuHasAVX2)) {
1038
4.48k
    InterpolateRow = InterpolateRow_Any_AVX2;
1039
4.48k
    if (IS_ALIGNED(src_width, 32)) {
1040
390
      InterpolateRow = InterpolateRow_AVX2;
1041
390
    }
1042
4.48k
  }
1043
4.48k
#endif
1044
#if defined(HAS_INTERPOLATEROW_NEON)
1045
  if (TestCpuFlag(kCpuHasNEON)) {
1046
    InterpolateRow = InterpolateRow_Any_NEON;
1047
    if (IS_ALIGNED(src_width, 16)) {
1048
      InterpolateRow = InterpolateRow_NEON;
1049
    }
1050
  }
1051
#endif
1052
#if defined(HAS_INTERPOLATEROW_SVE2)
1053
  if (TestCpuFlag(kCpuHasSVE2)) {
1054
    InterpolateRow = InterpolateRow_SVE2;
1055
  }
1056
#endif
1057
#if defined(HAS_INTERPOLATEROW_SME)
1058
  if (TestCpuFlag(kCpuHasSME)) {
1059
    InterpolateRow = InterpolateRow_SME;
1060
  }
1061
#endif
1062
#if defined(HAS_INTERPOLATEROW_LSX)
1063
  if (TestCpuFlag(kCpuHasLSX)) {
1064
    InterpolateRow = InterpolateRow_Any_LSX;
1065
    if (IS_ALIGNED(src_width, 32)) {
1066
      InterpolateRow = InterpolateRow_LSX;
1067
    }
1068
  }
1069
#endif
1070
#if defined(HAS_INTERPOLATEROW_RVV)
1071
  if (TestCpuFlag(kCpuHasRVV)) {
1072
    InterpolateRow = InterpolateRow_RVV;
1073
  }
1074
#endif
1075
1076
4.48k
#if defined(HAS_SCALEFILTERCOLS_SSSE3)
1077
4.48k
  if (TestCpuFlag(kCpuHasSSSE3) && src_width < 32768) {
1078
4.48k
    ScaleFilterCols = ScaleFilterCols_SSSE3;
1079
4.48k
  }
1080
4.48k
#endif
1081
#if defined(HAS_SCALEFILTERCOLS_NEON)
1082
  if (TestCpuFlag(kCpuHasNEON) && src_width < 32768) {
1083
    ScaleFilterCols = ScaleFilterCols_Any_NEON;
1084
    if (IS_ALIGNED(dst_width, 8)) {
1085
      ScaleFilterCols = ScaleFilterCols_NEON;
1086
    }
1087
  }
1088
#endif
1089
#if defined(HAS_SCALEFILTERCOLS_LSX)
1090
  if (TestCpuFlag(kCpuHasLSX) && src_width < 32768) {
1091
    ScaleFilterCols = ScaleFilterCols_Any_LSX;
1092
    if (IS_ALIGNED(dst_width, 16)) {
1093
      ScaleFilterCols = ScaleFilterCols_LSX;
1094
    }
1095
  }
1096
#endif
1097
4.48k
  if (y > max_y) {
1098
76
    y = max_y;
1099
76
  }
1100
1101
200k
  for (j = 0; j < dst_height; ++j) {
1102
195k
    int yi = y >> 16;
1103
195k
    const uint8_t* src = src_ptr + yi * src_stride;
1104
195k
    if (filtering == kFilterLinear) {
1105
110k
      ScaleFilterCols(dst_ptr, src, dst_width, x, dx);
1106
110k
    } else {
1107
85.5k
      int yf = (y >> 8) & 255;
1108
85.5k
      InterpolateRow(row, src, src_stride, src_width, yf);
1109
85.5k
      ScaleFilterCols(dst_ptr, row, dst_width, x, dx);
1110
85.5k
    }
1111
195k
    dst_ptr += dst_stride;
1112
195k
    y += dy;
1113
195k
    if (y > max_y) {
1114
6.00k
      y = max_y;
1115
6.00k
    }
1116
195k
  }
1117
4.48k
  free_aligned_buffer_64(row);
1118
4.48k
  return 0;
1119
4.48k
}
1120
1121
static int ScalePlaneBilinearDown_16(int src_width,
1122
                                     int src_height,
1123
                                     int dst_width,
1124
                                     int dst_height,
1125
                                     ptrdiff_t src_stride,
1126
                                     ptrdiff_t dst_stride,
1127
                                     const uint16_t* src_ptr,
1128
                                     uint16_t* dst_ptr,
1129
6.98k
                                     enum FilterMode filtering) {
1130
  // Initial source x/y coordinate and step values as 16.16 fixed point.
1131
6.98k
  int x = 0;
1132
6.98k
  int y = 0;
1133
6.98k
  int dx = 0;
1134
6.98k
  int dy = 0;
1135
  // TODO(fbarchard): Consider not allocating row buffer for kFilterLinear.
1136
  // Allocate a row buffer.
1137
6.98k
  align_buffer_64(row, src_width * 2);
1138
6.98k
  if (!row)
1139
0
    return 1;
1140
1141
6.98k
  const int max_y = (src_height - 1) << 16;
1142
6.98k
  int j;
1143
6.98k
  void (*ScaleFilterCols)(uint16_t* dst_ptr, const uint16_t* src_ptr,
1144
6.98k
                          int dst_width, int x, int dx) =
1145
6.98k
      (src_width >= 32768) ? ScaleFilterCols64_16_C : ScaleFilterCols_16_C;
1146
6.98k
  void (*InterpolateRow)(uint16_t* dst_ptr, const uint16_t* src_ptr,
1147
6.98k
                         ptrdiff_t src_stride, int dst_width,
1148
6.98k
                         int source_y_fraction) = InterpolateRow_16_C;
1149
6.98k
  ScaleSlope(src_width, src_height, dst_width, dst_height, filtering, &x, &y,
1150
6.98k
             &dx, &dy);
1151
6.98k
  src_width = Abs(src_width);
1152
1153
#if defined(HAS_INTERPOLATEROW_16_SSSE3)
1154
  if (TestCpuFlag(kCpuHasSSSE3)) {
1155
    InterpolateRow = InterpolateRow_16_Any_SSSE3;
1156
    if (IS_ALIGNED(src_width, 16)) {
1157
      InterpolateRow = InterpolateRow_16_SSSE3;
1158
    }
1159
  }
1160
#endif
1161
6.98k
#if defined(HAS_INTERPOLATEROW_16_AVX2)
1162
6.98k
  if (TestCpuFlag(kCpuHasAVX2)) {
1163
6.98k
    InterpolateRow = InterpolateRow_16_Any_AVX2;
1164
6.98k
    if (IS_ALIGNED(src_width, 32)) {
1165
794
      InterpolateRow = InterpolateRow_16_AVX2;
1166
794
    }
1167
6.98k
  }
1168
6.98k
#endif
1169
#if defined(HAS_INTERPOLATEROW_16_NEON)
1170
  if (TestCpuFlag(kCpuHasNEON)) {
1171
    InterpolateRow = InterpolateRow_16_Any_NEON;
1172
    if (IS_ALIGNED(src_width, 16)) {
1173
      InterpolateRow = InterpolateRow_16_NEON;
1174
    }
1175
  }
1176
#endif
1177
#if defined(HAS_INTERPOLATEROW_16_SME)
1178
  if (TestCpuFlag(kCpuHasSME)) {
1179
    InterpolateRow = InterpolateRow_16_SME;
1180
  }
1181
#endif
1182
1183
#if defined(HAS_SCALEFILTERCOLS_16_SSSE3)
1184
  if (TestCpuFlag(kCpuHasSSSE3) && src_width < 32768) {
1185
    ScaleFilterCols = ScaleFilterCols_16_SSSE3;
1186
  }
1187
#endif
1188
6.98k
  if (y > max_y) {
1189
41
    y = max_y;
1190
41
  }
1191
1192
469k
  for (j = 0; j < dst_height; ++j) {
1193
462k
    int yi = y >> 16;
1194
462k
    const uint16_t* src = src_ptr + yi * src_stride;
1195
462k
    if (filtering == kFilterLinear) {
1196
169k
      ScaleFilterCols(dst_ptr, src, dst_width, x, dx);
1197
293k
    } else {
1198
293k
      int yf = (y >> 8) & 255;
1199
293k
      InterpolateRow((uint16_t*)row, src, src_stride, src_width, yf);
1200
293k
      ScaleFilterCols(dst_ptr, (uint16_t*)row, dst_width, x, dx);
1201
293k
    }
1202
462k
    dst_ptr += dst_stride;
1203
462k
    y += dy;
1204
462k
    if (y > max_y) {
1205
8.22k
      y = max_y;
1206
8.22k
    }
1207
462k
  }
1208
6.98k
  free_aligned_buffer_64(row);
1209
6.98k
  return 0;
1210
6.98k
}
1211
1212
// Scale up down with bilinear interpolation.
1213
static int ScalePlaneBilinearUp(int src_width,
1214
                                int src_height,
1215
                                int dst_width,
1216
                                int dst_height,
1217
                                ptrdiff_t src_stride,
1218
                                ptrdiff_t dst_stride,
1219
                                const uint8_t* src_ptr,
1220
                                uint8_t* dst_ptr,
1221
5.42k
                                enum FilterMode filtering) {
1222
5.42k
  assert(src_width > 0);
1223
5.42k
  assert(src_height > 0);
1224
5.42k
  assert(dst_width > 0);
1225
5.42k
  assert(dst_height > 0);
1226
1227
5.42k
  int j;
1228
  // Initial source x/y coordinate and step values as 16.16 fixed point.
1229
5.42k
  int x = 0;
1230
5.42k
  int y = 0;
1231
5.42k
  int dx = 0;
1232
5.42k
  int dy = 0;
1233
5.42k
  const int max_y = (src_height - 1) << 16;
1234
5.42k
  void (*InterpolateRow)(uint8_t* dst_ptr, const uint8_t* src_ptr,
1235
5.42k
                         ptrdiff_t src_stride, int dst_width,
1236
5.42k
                         int source_y_fraction) = InterpolateRow_C;
1237
5.42k
  void (*ScaleFilterCols)(uint8_t* dst_ptr, const uint8_t* src_ptr,
1238
5.42k
                          int dst_width, int x, int dx) =
1239
5.42k
      filtering ? ScaleFilterCols_C : ScaleCols_C;
1240
5.42k
  ScaleSlope(src_width, src_height, dst_width, dst_height, filtering, &x, &y,
1241
5.42k
             &dx, &dy);
1242
5.42k
  assert(dy <= 65536);
1243
5.42k
  src_width = Abs(src_width);
1244
1245
5.42k
#if defined(HAS_INTERPOLATEROW_AVX2)
1246
5.42k
  if (TestCpuFlag(kCpuHasAVX2)) {
1247
5.42k
    InterpolateRow = InterpolateRow_Any_AVX2;
1248
5.42k
    if (IS_ALIGNED(dst_width, 32)) {
1249
2.53k
      InterpolateRow = InterpolateRow_AVX2;
1250
2.53k
    }
1251
5.42k
  }
1252
5.42k
#endif
1253
#if defined(HAS_INTERPOLATEROW_NEON)
1254
  if (TestCpuFlag(kCpuHasNEON)) {
1255
    InterpolateRow = InterpolateRow_Any_NEON;
1256
    if (IS_ALIGNED(dst_width, 16)) {
1257
      InterpolateRow = InterpolateRow_NEON;
1258
    }
1259
  }
1260
#endif
1261
#if defined(HAS_INTERPOLATEROW_SVE2)
1262
  if (TestCpuFlag(kCpuHasSVE2)) {
1263
    InterpolateRow = InterpolateRow_SVE2;
1264
  }
1265
#endif
1266
#if defined(HAS_INTERPOLATEROW_SME)
1267
  if (TestCpuFlag(kCpuHasSME)) {
1268
    InterpolateRow = InterpolateRow_SME;
1269
  }
1270
#endif
1271
#if defined(HAS_INTERPOLATEROW_RVV)
1272
  if (TestCpuFlag(kCpuHasRVV)) {
1273
    InterpolateRow = InterpolateRow_RVV;
1274
  }
1275
#endif
1276
1277
5.42k
  if (filtering && src_width >= 32768) {
1278
0
    ScaleFilterCols = ScaleFilterCols64_C;
1279
0
  }
1280
5.42k
#if defined(HAS_SCALEFILTERCOLS_SSSE3)
1281
5.42k
  if (filtering && TestCpuFlag(kCpuHasSSSE3) && src_width < 32768) {
1282
5.42k
    ScaleFilterCols = ScaleFilterCols_SSSE3;
1283
5.42k
  }
1284
5.42k
#endif
1285
#if defined(HAS_SCALEFILTERCOLS_NEON)
1286
  if (filtering && TestCpuFlag(kCpuHasNEON) && src_width < 32768) {
1287
    ScaleFilterCols = ScaleFilterCols_Any_NEON;
1288
    if (IS_ALIGNED(dst_width, 8)) {
1289
      ScaleFilterCols = ScaleFilterCols_NEON;
1290
    }
1291
  }
1292
#endif
1293
#if defined(HAS_SCALEFILTERCOLS_LSX)
1294
  if (filtering && TestCpuFlag(kCpuHasLSX) && src_width < 32768) {
1295
    ScaleFilterCols = ScaleFilterCols_Any_LSX;
1296
    if (IS_ALIGNED(dst_width, 16)) {
1297
      ScaleFilterCols = ScaleFilterCols_LSX;
1298
    }
1299
  }
1300
#endif
1301
5.42k
  if (!filtering && src_width * 2 == dst_width && x < 0x8000) {
1302
0
    ScaleFilterCols = ScaleColsUp2_C;
1303
#if defined(HAS_SCALECOLS_SSE2)
1304
    if (TestCpuFlag(kCpuHasSSE2) && IS_ALIGNED(dst_width, 8)) {
1305
      ScaleFilterCols = ScaleColsUp2_SSE2;
1306
    }
1307
#endif
1308
0
  }
1309
1310
5.42k
  if (y > max_y) {
1311
1.18k
    y = max_y;
1312
1.18k
  }
1313
5.42k
  {
1314
5.42k
    int yi = y >> 16;
1315
5.42k
    const uint8_t* src = src_ptr + yi * src_stride;
1316
1317
    // Allocate 2 row buffers.
1318
5.42k
    const int row_size = (dst_width + 31) & ~31;
1319
5.42k
    align_buffer_64(row, row_size * 2);
1320
5.42k
    if (!row)
1321
0
      return 1;
1322
1323
5.42k
    uint8_t* rowptr = row;
1324
5.42k
    ptrdiff_t rowstride = row_size;
1325
5.42k
    int lasty = yi;
1326
1327
5.42k
    ScaleFilterCols(rowptr, src, dst_width, x, dx);
1328
5.42k
    if (src_height > 1) {
1329
4.24k
      src += src_stride;
1330
4.24k
    }
1331
5.42k
    ScaleFilterCols(rowptr + rowstride, src, dst_width, x, dx);
1332
5.42k
    if (src_height > 2) {
1333
3.97k
      src += src_stride;
1334
3.97k
    }
1335
1336
    // 2-row rolling buffer:
1337
    // rowptr and (rowptr + rowstride) hold the scaled rows for yi and yi + 1.
1338
    // Because dy <= 65536 (dy <= 1.0 in 16.16), yi advances in unit steps.
1339
    // When yi != lasty:
1340
    // 1. Scale the next source row into the older buffer (rowptr).
1341
    // 2. Swap buffer pointers (rowptr += rowstride; rowstride = -rowstride;)
1342
    //    so rowptr points to yi and (rowptr + rowstride) points to yi + 1.
1343
    // 3. Advance src by 1 row if row yi + 2 exists ((y + 65536) < max_y),
1344
    //    otherwise clamp src at (src_height - 1) to avoid reading out of bounds.
1345
2.15M
    for (j = 0; j < dst_height; ++j) {
1346
2.14M
      if (y > max_y) {
1347
261k
        y = max_y;
1348
261k
      }
1349
2.14M
      yi = y >> 16;
1350
2.14M
      if (yi != lasty) {
1351
123k
        ScaleFilterCols(rowptr, src, dst_width, x, dx);
1352
123k
        rowptr += rowstride;
1353
123k
        rowstride = -rowstride;
1354
123k
        lasty = yi;
1355
123k
        if ((y + 65536) < max_y) {
1356
119k
          src += src_stride;
1357
119k
        }
1358
123k
      }
1359
2.14M
      if (filtering == kFilterLinear) {
1360
263k
        InterpolateRow(dst_ptr, rowptr, 0, dst_width, 0);
1361
1.88M
      } else {
1362
1.88M
        int yf = (y >> 8) & 255;
1363
1.88M
        InterpolateRow(dst_ptr, rowptr, rowstride, dst_width, yf);
1364
1.88M
      }
1365
2.14M
      dst_ptr += dst_stride;
1366
2.14M
      y += dy;
1367
2.14M
    }
1368
5.42k
    free_aligned_buffer_64(row);
1369
5.42k
  }
1370
0
  return 0;
1371
5.42k
}
1372
1373
// Scale plane, horizontally up by 2 times.
1374
// Uses linear filter horizontally, nearest vertically.
1375
// This is an optimized version for scaling up a plane to 2 times of
1376
// its original width, using linear interpolation.
1377
// This is used to scale U and V planes of I422 to I444.
1378
static void ScalePlaneUp2_Linear(int src_width,
1379
                                 int src_height,
1380
                                 int dst_width,
1381
                                 int dst_height,
1382
                                 ptrdiff_t src_stride,
1383
                                 ptrdiff_t dst_stride,
1384
                                 const uint8_t* src_ptr,
1385
290
                                 uint8_t* dst_ptr) {
1386
290
  void (*ScaleRowUp)(const uint8_t* src_ptr, uint8_t* dst_ptr, int dst_width) =
1387
290
      ScaleRowUp2_Linear_Any_C;
1388
290
  int i;
1389
290
  int y;
1390
290
  int dy;
1391
1392
290
  (void)src_width;
1393
  // This function can only scale up by 2 times horizontally.
1394
290
  assert(src_width == ((dst_width + 1) / 2));
1395
1396
290
#ifdef HAS_SCALEROWUP2_LINEAR_SSE2
1397
290
  if (TestCpuFlag(kCpuHasSSE2)) {
1398
290
    ScaleRowUp = ScaleRowUp2_Linear_Any_SSE2;
1399
290
  }
1400
290
#endif
1401
1402
290
#ifdef HAS_SCALEROWUP2_LINEAR_SSSE3
1403
290
  if (TestCpuFlag(kCpuHasSSSE3)) {
1404
290
    ScaleRowUp = ScaleRowUp2_Linear_Any_SSSE3;
1405
290
  }
1406
290
#endif
1407
1408
290
#ifdef HAS_SCALEROWUP2_LINEAR_AVX2
1409
290
  if (TestCpuFlag(kCpuHasAVX2)) {
1410
290
    ScaleRowUp = ScaleRowUp2_Linear_Any_AVX2;
1411
290
  }
1412
290
#endif
1413
1414
#ifdef HAS_SCALEROWUP2_LINEAR_NEON
1415
  if (TestCpuFlag(kCpuHasNEON)) {
1416
    ScaleRowUp = ScaleRowUp2_Linear_Any_NEON;
1417
  }
1418
#endif
1419
#ifdef HAS_SCALEROWUP2_LINEAR_RVV
1420
  if (TestCpuFlag(kCpuHasRVV)) {
1421
    ScaleRowUp = ScaleRowUp2_Linear_RVV;
1422
  }
1423
#endif
1424
1425
290
  if (dst_height == 1) {
1426
31
    ScaleRowUp(src_ptr + ((src_height - 1) / 2) * src_stride, dst_ptr,
1427
31
               dst_width);
1428
259
  } else {
1429
259
    dy = FixedDiv(src_height - 1, dst_height - 1);
1430
259
    y = (1 << 15) - 1;
1431
316k
    for (i = 0; i < dst_height; ++i) {
1432
316k
      ScaleRowUp(src_ptr + (y >> 16) * src_stride, dst_ptr, dst_width);
1433
316k
      dst_ptr += dst_stride;
1434
316k
      y += dy;
1435
316k
    }
1436
259
  }
1437
290
}
1438
1439
// Scale plane, up by 2 times.
1440
// This is an optimized version for scaling up a plane to 2 times of
1441
// its original size, using bilinear interpolation.
1442
// This is used to scale U and V planes of I420 to I444.
1443
static void ScalePlaneUp2_Bilinear(int src_width,
1444
                                   int src_height,
1445
                                   int dst_width,
1446
                                   int dst_height,
1447
                                   ptrdiff_t src_stride,
1448
                                   ptrdiff_t dst_stride,
1449
                                   const uint8_t* src_ptr,
1450
235
                                   uint8_t* dst_ptr) {
1451
235
  void (*Scale2RowUp)(const uint8_t* src_ptr, ptrdiff_t src_stride,
1452
235
                      uint8_t* dst_ptr, ptrdiff_t dst_stride, int dst_width) =
1453
235
      ScaleRowUp2_Bilinear_Any_C;
1454
235
  int x;
1455
1456
235
  (void)src_width;
1457
  // This function can only scale up by 2 times.
1458
235
  assert(src_width == ((dst_width + 1) / 2));
1459
235
  assert(src_height == ((dst_height + 1) / 2));
1460
1461
235
#ifdef HAS_SCALEROWUP2_BILINEAR_SSE2
1462
235
  if (TestCpuFlag(kCpuHasSSE2)) {
1463
235
    Scale2RowUp = ScaleRowUp2_Bilinear_Any_SSE2;
1464
235
  }
1465
235
#endif
1466
1467
235
#ifdef HAS_SCALEROWUP2_BILINEAR_SSSE3
1468
235
  if (TestCpuFlag(kCpuHasSSSE3)) {
1469
235
    Scale2RowUp = ScaleRowUp2_Bilinear_Any_SSSE3;
1470
235
  }
1471
235
#endif
1472
1473
235
#ifdef HAS_SCALEROWUP2_BILINEAR_AVX2
1474
235
  if (TestCpuFlag(kCpuHasAVX2)) {
1475
235
    Scale2RowUp = ScaleRowUp2_Bilinear_Any_AVX2;
1476
235
  }
1477
235
#endif
1478
1479
#ifdef HAS_SCALEROWUP2_BILINEAR_NEON
1480
  if (TestCpuFlag(kCpuHasNEON)) {
1481
    Scale2RowUp = ScaleRowUp2_Bilinear_Any_NEON;
1482
  }
1483
#endif
1484
#ifdef HAS_SCALEROWUP2_BILINEAR_RVV
1485
  if (TestCpuFlag(kCpuHasRVV)) {
1486
    Scale2RowUp = ScaleRowUp2_Bilinear_RVV;
1487
  }
1488
#endif
1489
1490
235
  Scale2RowUp(src_ptr, 0, dst_ptr, 0, dst_width);
1491
235
  dst_ptr += dst_stride;
1492
8.40k
  for (x = 0; x < src_height - 1; ++x) {
1493
8.17k
    Scale2RowUp(src_ptr, src_stride, dst_ptr, dst_stride, dst_width);
1494
8.17k
    src_ptr += src_stride;
1495
    // TODO(fbarchard): Test performance of writing one row of destination at a
1496
    // time.
1497
8.17k
    dst_ptr += 2 * dst_stride;
1498
8.17k
  }
1499
235
  if (!(dst_height & 1)) {
1500
172
    Scale2RowUp(src_ptr, 0, dst_ptr, 0, dst_width);
1501
172
  }
1502
235
}
1503
1504
// Scale at most 14 bit plane, horizontally up by 2 times.
1505
// This is an optimized version for scaling up a plane to 2 times of
1506
// its original width, using linear interpolation.
1507
// stride is in count of uint16_t.
1508
// This is used to scale U and V planes of I210 to I410 and I212 to I412.
1509
static void ScalePlaneUp2_12_Linear(int src_width,
1510
                                    int src_height,
1511
                                    int dst_width,
1512
                                    int dst_height,
1513
                                    ptrdiff_t src_stride,
1514
                                    ptrdiff_t dst_stride,
1515
                                    const uint16_t* src_ptr,
1516
283
                                    uint16_t* dst_ptr) {
1517
283
  void (*ScaleRowUp)(const uint16_t* src_ptr, uint16_t* dst_ptr,
1518
283
                     int dst_width) = ScaleRowUp2_Linear_16_Any_C;
1519
283
  int i;
1520
283
  int y;
1521
283
  int dy;
1522
1523
283
  (void)src_width;
1524
  // This function can only scale up by 2 times horizontally.
1525
283
  assert(src_width == ((dst_width + 1) / 2));
1526
1527
283
#ifdef HAS_SCALEROWUP2_LINEAR_12_SSSE3
1528
283
  if (TestCpuFlag(kCpuHasSSSE3)) {
1529
283
    ScaleRowUp = ScaleRowUp2_Linear_12_Any_SSSE3;
1530
283
  }
1531
283
#endif
1532
1533
283
#ifdef HAS_SCALEROWUP2_LINEAR_12_AVX2
1534
283
  if (TestCpuFlag(kCpuHasAVX2)) {
1535
283
    ScaleRowUp = ScaleRowUp2_Linear_12_Any_AVX2;
1536
283
  }
1537
283
#endif
1538
1539
#ifdef HAS_SCALEROWUP2_LINEAR_12_NEON
1540
  if (TestCpuFlag(kCpuHasNEON)) {
1541
    ScaleRowUp = ScaleRowUp2_Linear_12_Any_NEON;
1542
  }
1543
#endif
1544
1545
283
  if (dst_height == 1) {
1546
18
    ScaleRowUp(src_ptr + ((src_height - 1) / 2) * src_stride, dst_ptr,
1547
18
               dst_width);
1548
265
  } else {
1549
265
    dy = FixedDiv(src_height - 1, dst_height - 1);
1550
265
    y = (1 << 15) - 1;
1551
620k
    for (i = 0; i < dst_height; ++i) {
1552
620k
      ScaleRowUp(src_ptr + (y >> 16) * src_stride, dst_ptr, dst_width);
1553
620k
      dst_ptr += dst_stride;
1554
620k
      y += dy;
1555
620k
    }
1556
265
  }
1557
283
}
1558
1559
// Scale at most 12 bit plane, up by 2 times.
1560
// This is an optimized version for scaling up a plane to 2 times of
1561
// its original size, using bilinear interpolation.
1562
// stride is in count of uint16_t.
1563
// This is used to scale U and V planes of I010 to I410 and I012 to I412.
1564
static void ScalePlaneUp2_12_Bilinear(int src_width,
1565
                                      int src_height,
1566
                                      int dst_width,
1567
                                      int dst_height,
1568
                                      ptrdiff_t src_stride,
1569
                                      ptrdiff_t dst_stride,
1570
                                      const uint16_t* src_ptr,
1571
178
                                      uint16_t* dst_ptr) {
1572
178
  void (*Scale2RowUp)(const uint16_t* src_ptr, ptrdiff_t src_stride,
1573
178
                      uint16_t* dst_ptr, ptrdiff_t dst_stride, int dst_width) =
1574
178
      ScaleRowUp2_Bilinear_16_Any_C;
1575
178
  int x;
1576
1577
178
  (void)src_width;
1578
  // This function can only scale up by 2 times.
1579
178
  assert(src_width == ((dst_width + 1) / 2));
1580
178
  assert(src_height == ((dst_height + 1) / 2));
1581
1582
178
#ifdef HAS_SCALEROWUP2_BILINEAR_12_SSSE3
1583
178
  if (TestCpuFlag(kCpuHasSSSE3)) {
1584
178
    Scale2RowUp = ScaleRowUp2_Bilinear_12_Any_SSSE3;
1585
178
  }
1586
178
#endif
1587
1588
178
#ifdef HAS_SCALEROWUP2_BILINEAR_12_AVX2
1589
178
  if (TestCpuFlag(kCpuHasAVX2)) {
1590
178
    Scale2RowUp = ScaleRowUp2_Bilinear_12_Any_AVX2;
1591
178
  }
1592
178
#endif
1593
1594
#ifdef HAS_SCALEROWUP2_BILINEAR_12_NEON
1595
  if (TestCpuFlag(kCpuHasNEON)) {
1596
    Scale2RowUp = ScaleRowUp2_Bilinear_12_Any_NEON;
1597
  }
1598
#endif
1599
1600
178
  Scale2RowUp(src_ptr, 0, dst_ptr, 0, dst_width);
1601
178
  dst_ptr += dst_stride;
1602
6.77k
  for (x = 0; x < src_height - 1; ++x) {
1603
6.59k
    Scale2RowUp(src_ptr, src_stride, dst_ptr, dst_stride, dst_width);
1604
6.59k
    src_ptr += src_stride;
1605
6.59k
    dst_ptr += 2 * dst_stride;
1606
6.59k
  }
1607
178
  if (!(dst_height & 1)) {
1608
113
    Scale2RowUp(src_ptr, 0, dst_ptr, 0, dst_width);
1609
113
  }
1610
178
}
1611
1612
static void ScalePlaneUp2_16_Linear(int src_width,
1613
                                    int src_height,
1614
                                    int dst_width,
1615
                                    int dst_height,
1616
                                    ptrdiff_t src_stride,
1617
                                    ptrdiff_t dst_stride,
1618
                                    const uint16_t* src_ptr,
1619
0
                                    uint16_t* dst_ptr) {
1620
0
  void (*ScaleRowUp)(const uint16_t* src_ptr, uint16_t* dst_ptr,
1621
0
                     int dst_width) = ScaleRowUp2_Linear_16_Any_C;
1622
0
  int i;
1623
0
  int y;
1624
0
  int dy;
1625
1626
0
  (void)src_width;
1627
  // This function can only scale up by 2 times horizontally.
1628
0
  assert(src_width == ((dst_width + 1) / 2));
1629
1630
0
#ifdef HAS_SCALEROWUP2_LINEAR_16_SSE2
1631
0
  if (TestCpuFlag(kCpuHasSSE2)) {
1632
0
    ScaleRowUp = ScaleRowUp2_Linear_16_Any_SSE2;
1633
0
  }
1634
0
#endif
1635
1636
0
#ifdef HAS_SCALEROWUP2_LINEAR_16_AVX2
1637
0
  if (TestCpuFlag(kCpuHasAVX2)) {
1638
0
    ScaleRowUp = ScaleRowUp2_Linear_16_Any_AVX2;
1639
0
  }
1640
0
#endif
1641
1642
#ifdef HAS_SCALEROWUP2_LINEAR_16_NEON
1643
  if (TestCpuFlag(kCpuHasNEON)) {
1644
    ScaleRowUp = ScaleRowUp2_Linear_16_Any_NEON;
1645
  }
1646
#endif
1647
1648
0
  if (dst_height == 1) {
1649
0
    ScaleRowUp(src_ptr + ((src_height - 1) / 2) * src_stride, dst_ptr,
1650
0
               dst_width);
1651
0
  } else {
1652
0
    dy = FixedDiv(src_height - 1, dst_height - 1);
1653
0
    y = (1 << 15) - 1;
1654
0
    for (i = 0; i < dst_height; ++i) {
1655
0
      ScaleRowUp(src_ptr + (y >> 16) * src_stride, dst_ptr, dst_width);
1656
0
      dst_ptr += dst_stride;
1657
0
      y += dy;
1658
0
    }
1659
0
  }
1660
0
}
1661
1662
static void ScalePlaneUp2_16_Bilinear(int src_width,
1663
                                      int src_height,
1664
                                      int dst_width,
1665
                                      int dst_height,
1666
                                      ptrdiff_t src_stride,
1667
                                      ptrdiff_t dst_stride,
1668
                                      const uint16_t* src_ptr,
1669
0
                                      uint16_t* dst_ptr) {
1670
0
  void (*Scale2RowUp)(const uint16_t* src_ptr, ptrdiff_t src_stride,
1671
0
                      uint16_t* dst_ptr, ptrdiff_t dst_stride, int dst_width) =
1672
0
      ScaleRowUp2_Bilinear_16_Any_C;
1673
0
  int x;
1674
1675
0
  (void)src_width;
1676
  // This function can only scale up by 2 times.
1677
0
  assert(src_width == ((dst_width + 1) / 2));
1678
0
  assert(src_height == ((dst_height + 1) / 2));
1679
1680
0
#ifdef HAS_SCALEROWUP2_BILINEAR_16_SSE2
1681
0
  if (TestCpuFlag(kCpuHasSSE2)) {
1682
0
    Scale2RowUp = ScaleRowUp2_Bilinear_16_Any_SSE2;
1683
0
  }
1684
0
#endif
1685
1686
0
#ifdef HAS_SCALEROWUP2_BILINEAR_16_AVX2
1687
0
  if (TestCpuFlag(kCpuHasAVX2)) {
1688
0
    Scale2RowUp = ScaleRowUp2_Bilinear_16_Any_AVX2;
1689
0
  }
1690
0
#endif
1691
1692
#ifdef HAS_SCALEROWUP2_BILINEAR_16_NEON
1693
  if (TestCpuFlag(kCpuHasNEON)) {
1694
    Scale2RowUp = ScaleRowUp2_Bilinear_16_Any_NEON;
1695
  }
1696
#endif
1697
1698
0
  Scale2RowUp(src_ptr, 0, dst_ptr, 0, dst_width);
1699
0
  dst_ptr += dst_stride;
1700
0
  for (x = 0; x < src_height - 1; ++x) {
1701
0
    Scale2RowUp(src_ptr, src_stride, dst_ptr, dst_stride, dst_width);
1702
0
    src_ptr += src_stride;
1703
0
    dst_ptr += 2 * dst_stride;
1704
0
  }
1705
0
  if (!(dst_height & 1)) {
1706
0
    Scale2RowUp(src_ptr, 0, dst_ptr, 0, dst_width);
1707
0
  }
1708
0
}
1709
1710
static int ScalePlaneBilinearUp_16(int src_width,
1711
                                   int src_height,
1712
                                   int dst_width,
1713
                                   int dst_height,
1714
                                   ptrdiff_t src_stride,
1715
                                   ptrdiff_t dst_stride,
1716
                                   const uint16_t* src_ptr,
1717
                                   uint16_t* dst_ptr,
1718
5.85k
                                   enum FilterMode filtering) {
1719
5.85k
  assert(src_width > 0);
1720
5.85k
  assert(src_height > 0);
1721
5.85k
  assert(dst_width > 0);
1722
5.85k
  assert(dst_height > 0);
1723
1724
5.85k
  int j;
1725
  // Initial source x/y coordinate and step values as 16.16 fixed point.
1726
5.85k
  int x = 0;
1727
5.85k
  int y = 0;
1728
5.85k
  int dx = 0;
1729
5.85k
  int dy = 0;
1730
5.85k
  const int max_y = (src_height - 1) << 16;
1731
5.85k
  void (*InterpolateRow)(uint16_t* dst_ptr, const uint16_t* src_ptr,
1732
5.85k
                         ptrdiff_t src_stride, int dst_width,
1733
5.85k
                         int source_y_fraction) = InterpolateRow_16_C;
1734
5.85k
  void (*ScaleFilterCols)(uint16_t* dst_ptr, const uint16_t* src_ptr,
1735
5.85k
                          int dst_width, int x, int dx) =
1736
5.85k
      filtering ? ScaleFilterCols_16_C : ScaleCols_16_C;
1737
5.85k
  ScaleSlope(src_width, src_height, dst_width, dst_height, filtering, &x, &y,
1738
5.85k
             &dx, &dy);
1739
5.85k
  assert(dy <= 65536);
1740
5.85k
  src_width = Abs(src_width);
1741
1742
#if defined(HAS_INTERPOLATEROW_16_SSSE3)
1743
  if (TestCpuFlag(kCpuHasSSSE3)) {
1744
    InterpolateRow = InterpolateRow_16_Any_SSSE3;
1745
    if (IS_ALIGNED(dst_width, 16)) {
1746
      InterpolateRow = InterpolateRow_16_SSSE3;
1747
    }
1748
  }
1749
#endif
1750
5.85k
#if defined(HAS_INTERPOLATEROW_16_AVX2)
1751
5.85k
  if (TestCpuFlag(kCpuHasAVX2)) {
1752
5.85k
    InterpolateRow = InterpolateRow_16_Any_AVX2;
1753
5.85k
    if (IS_ALIGNED(dst_width, 32)) {
1754
2.44k
      InterpolateRow = InterpolateRow_16_AVX2;
1755
2.44k
    }
1756
5.85k
  }
1757
5.85k
#endif
1758
#if defined(HAS_INTERPOLATEROW_16_NEON)
1759
  if (TestCpuFlag(kCpuHasNEON)) {
1760
    InterpolateRow = InterpolateRow_16_Any_NEON;
1761
    if (IS_ALIGNED(dst_width, 16)) {
1762
      InterpolateRow = InterpolateRow_16_NEON;
1763
    }
1764
  }
1765
#endif
1766
#if defined(HAS_INTERPOLATEROW_16_SME)
1767
  if (TestCpuFlag(kCpuHasSME)) {
1768
    InterpolateRow = InterpolateRow_16_SME;
1769
  }
1770
#endif
1771
1772
5.85k
  if (filtering && src_width >= 32768) {
1773
0
    ScaleFilterCols = ScaleFilterCols64_16_C;
1774
0
  }
1775
#if defined(HAS_SCALEFILTERCOLS_16_SSSE3)
1776
  if (filtering && TestCpuFlag(kCpuHasSSSE3) && src_width < 32768) {
1777
    ScaleFilterCols = ScaleFilterCols_16_SSSE3;
1778
  }
1779
#endif
1780
5.85k
  if (!filtering && src_width * 2 == dst_width && x < 0x8000) {
1781
0
    ScaleFilterCols = ScaleColsUp2_16_C;
1782
#if defined(HAS_SCALECOLS_16_SSE2)
1783
    if (TestCpuFlag(kCpuHasSSE2) && IS_ALIGNED(dst_width, 8)) {
1784
      ScaleFilterCols = ScaleColsUp2_16_SSE2;
1785
    }
1786
#endif
1787
0
  }
1788
1789
5.85k
  if (y > max_y) {
1790
868
    y = max_y;
1791
868
  }
1792
5.85k
  {
1793
5.85k
    int yi = y >> 16;
1794
5.85k
    const uint16_t* src = src_ptr + yi * src_stride;
1795
1796
    // Allocate 2 row buffers.
1797
5.85k
    const int row_size = (dst_width + 31) & ~31;
1798
5.85k
    align_buffer_64(row, row_size * 4);
1799
5.85k
    ptrdiff_t rowstride = row_size;
1800
5.85k
    int lasty = yi;
1801
5.85k
    uint16_t* rowptr = (uint16_t*)row;
1802
5.85k
    if (!row)
1803
0
      return 1;
1804
1805
5.85k
    ScaleFilterCols(rowptr, src, dst_width, x, dx);
1806
5.85k
    if (src_height > 1) {
1807
4.98k
      src += src_stride;
1808
4.98k
    }
1809
5.85k
    ScaleFilterCols(rowptr + rowstride, src, dst_width, x, dx);
1810
5.85k
    if (src_height > 2) {
1811
4.05k
      src += src_stride;
1812
4.05k
    }
1813
1814
    // 2-row rolling buffer:
1815
    // rowptr and (rowptr + rowstride) hold the scaled rows for yi and yi + 1.
1816
    // Because dy <= 65536 (dy <= 1.0 in 16.16), yi advances in unit steps.
1817
    // When yi != lasty:
1818
    // 1. Scale the next source row into the older buffer (rowptr).
1819
    // 2. Swap buffer pointers (rowptr += rowstride; rowstride = -rowstride;)
1820
    //    so rowptr points to yi and (rowptr + rowstride) points to yi + 1.
1821
    // 3. Advance src by 1 row if row yi + 2 exists ((y + 65536) < max_y),
1822
    //    otherwise clamp src at (src_height - 1) to avoid reading out of bounds.
1823
3.58M
    for (j = 0; j < dst_height; ++j) {
1824
3.57M
      if (y > max_y) {
1825
323k
        y = max_y;
1826
323k
      }
1827
3.57M
      yi = y >> 16;
1828
3.57M
      if (yi != lasty) {
1829
159k
        ScaleFilterCols(rowptr, src, dst_width, x, dx);
1830
159k
        rowptr += rowstride;
1831
159k
        rowstride = -rowstride;
1832
159k
        lasty = yi;
1833
159k
        if ((y + 65536) < max_y) {
1834
155k
          src += src_stride;
1835
155k
        }
1836
159k
      }
1837
3.57M
      if (filtering == kFilterLinear) {
1838
323k
        InterpolateRow(dst_ptr, rowptr, 0, dst_width, 0);
1839
3.25M
      } else {
1840
3.25M
        int yf = (y >> 8) & 255;
1841
3.25M
        InterpolateRow(dst_ptr, rowptr, rowstride, dst_width, yf);
1842
3.25M
      }
1843
3.57M
      dst_ptr += dst_stride;
1844
3.57M
      y += dy;
1845
3.57M
    }
1846
5.85k
    free_aligned_buffer_64(row);
1847
5.85k
  }
1848
0
  return 0;
1849
5.85k
}
1850
1851
// Scale Plane to/from any dimensions, without interpolation.
1852
// Fixed point math is used for performance: The upper 16 bits
1853
// of x and dx is the integer part of the source position and
1854
// the lower 16 bits are the fixed decimal part.
1855
1856
static void ScalePlaneSimple(int src_width,
1857
                             int src_height,
1858
                             int dst_width,
1859
                             int dst_height,
1860
                             ptrdiff_t src_stride,
1861
                             ptrdiff_t dst_stride,
1862
                             const uint8_t* src_ptr,
1863
1.29k
                             uint8_t* dst_ptr) {
1864
1.29k
  int i;
1865
1.29k
  void (*ScaleCols)(uint8_t* dst_ptr, const uint8_t* src_ptr, int dst_width,
1866
1.29k
                    int x, int dx) = ScaleCols_C;
1867
  // Initial source x/y coordinate and step values as 16.16 fixed point.
1868
1.29k
  int x = 0;
1869
1.29k
  int y = 0;
1870
1.29k
  int dx = 0;
1871
1.29k
  int dy = 0;
1872
1.29k
  ScaleSlope(src_width, src_height, dst_width, dst_height, kFilterNone, &x, &y,
1873
1.29k
             &dx, &dy);
1874
1.29k
  src_width = Abs(src_width);
1875
1876
1.29k
  if (src_width * 2 == dst_width && x < 0x8000) {
1877
105
    ScaleCols = ScaleColsUp2_C;
1878
#if defined(HAS_SCALECOLS_SSE2)
1879
    if (TestCpuFlag(kCpuHasSSE2) && IS_ALIGNED(dst_width, 8)) {
1880
      ScaleCols = ScaleColsUp2_SSE2;
1881
    }
1882
#endif
1883
105
  }
1884
1885
781k
  for (i = 0; i < dst_height; ++i) {
1886
780k
    ScaleCols(dst_ptr, src_ptr + (y >> 16) * src_stride, dst_width, x, dx);
1887
780k
    dst_ptr += dst_stride;
1888
780k
    y += dy;
1889
780k
  }
1890
1.29k
}
1891
1892
static void ScalePlaneSimple_16(int src_width,
1893
                                int src_height,
1894
                                int dst_width,
1895
                                int dst_height,
1896
                                ptrdiff_t src_stride,
1897
                                ptrdiff_t dst_stride,
1898
                                const uint16_t* src_ptr,
1899
1.06k
                                uint16_t* dst_ptr) {
1900
1.06k
  int i;
1901
1.06k
  void (*ScaleCols)(uint16_t* dst_ptr, const uint16_t* src_ptr, int dst_width,
1902
1.06k
                    int x, int dx) = ScaleCols_16_C;
1903
  // Initial source x/y coordinate and step values as 16.16 fixed point.
1904
1.06k
  int x = 0;
1905
1.06k
  int y = 0;
1906
1.06k
  int dx = 0;
1907
1.06k
  int dy = 0;
1908
1.06k
  ScaleSlope(src_width, src_height, dst_width, dst_height, kFilterNone, &x, &y,
1909
1.06k
             &dx, &dy);
1910
1.06k
  src_width = Abs(src_width);
1911
1912
1.06k
  if (src_width * 2 == dst_width && x < 0x8000) {
1913
87
    ScaleCols = ScaleColsUp2_16_C;
1914
#if defined(HAS_SCALECOLS_16_SSE2)
1915
    if (TestCpuFlag(kCpuHasSSE2) && IS_ALIGNED(dst_width, 8)) {
1916
      ScaleCols = ScaleColsUp2_16_SSE2;
1917
    }
1918
#endif
1919
87
  }
1920
1921
899k
  for (i = 0; i < dst_height; ++i) {
1922
898k
    ScaleCols(dst_ptr, src_ptr + (y >> 16) * src_stride, dst_width, x, dx);
1923
898k
    dst_ptr += dst_stride;
1924
898k
    y += dy;
1925
898k
  }
1926
1.06k
}
1927
1928
// Scale a plane.
1929
// This function dispatches to a specialized scaler based on scale factor.
1930
LIBYUV_API
1931
int ScalePlane(const uint8_t* src,
1932
               int src_stride,
1933
               int src_width,
1934
               int src_height,
1935
               uint8_t* dst,
1936
               int dst_stride,
1937
               int dst_width,
1938
               int dst_height,
1939
15.4k
               enum FilterMode filtering) {
1940
  // Reject dimensions larger than 32768 (or smaller than -32768 for height).
1941
  // This prevents FixedDiv signed integer overflows that can lead to division
1942
  // by zero/overflow crashes (SIGFPE on x86) or incorrect step calculations.
1943
15.4k
  if (!src || src_width <= 0 || src_height == 0 || src_width > 32768 ||
1944
15.4k
      src_height < -32768 || src_height > 32768 || !dst || dst_width <= 0 ||
1945
15.4k
      dst_height <= 0) {
1946
0
    return -1;
1947
0
  }
1948
  // Simplify filtering when possible.
1949
15.4k
  filtering = ScaleFilterReduce(src_width, src_height, dst_width, dst_height,
1950
15.4k
                                filtering);
1951
1952
  // Negative height means invert the image.
1953
15.4k
  if (src_height < 0) {
1954
0
    src_height = -src_height;
1955
0
    src = src + (src_height - 1) * (ptrdiff_t)src_stride;
1956
0
    src_stride = -src_stride;
1957
0
  }
1958
  // Use specialized scales to improve performance for common resolutions.
1959
  // For example, all the 1/2 scalings will use ScalePlaneDown2()
1960
15.4k
  if (dst_width == src_width && dst_height == src_height) {
1961
    // Straight copy.
1962
258
    CopyPlane(src, src_stride, dst, dst_stride, dst_width, dst_height);
1963
258
    return 0;
1964
258
  }
1965
15.1k
  if (dst_width == src_width && filtering != kFilterBox) {
1966
2.13k
    int dy = 0;
1967
2.13k
    int y = 0;
1968
    // When scaling down, use the center 2 rows to filter.
1969
    // When scaling up, last row of destination uses the last 2 source rows.
1970
2.13k
    if (dst_height <= src_height) {
1971
728
      dy = FixedDiv(src_height, dst_height);
1972
728
      y = CENTERSTART(dy, -32768);  // Subtract 0.5 (32768) to center filter.
1973
1.40k
    } else if (src_height > 1 && dst_height > 1) {
1974
1.26k
      dy = FixedDiv1(src_height, dst_height);
1975
1.26k
    }
1976
    // Arbitrary scale vertically, but unscaled horizontally.
1977
2.13k
    ScalePlaneVertical(src_height, dst_width, dst_height, src_stride,
1978
2.13k
                       dst_stride, src, dst, 0, y, dy, /*bpp=*/1, filtering);
1979
2.13k
    return 0;
1980
2.13k
  }
1981
13.0k
  if (dst_width <= Abs(src_width) && dst_height <= src_height) {
1982
    // Scale down.
1983
2.76k
    if (4 * dst_width == 3 * src_width && 4 * dst_height == 3 * src_height) {
1984
      // optimized, 3/4
1985
2
      ScalePlaneDown34(src_width, src_height, dst_width, dst_height, src_stride,
1986
2
                       dst_stride, src, dst, filtering);
1987
2
      return 0;
1988
2
    }
1989
2.76k
    if (2 * dst_width == src_width && 2 * dst_height == src_height) {
1990
      // optimized, 1/2
1991
105
      ScalePlaneDown2(src_width, src_height, dst_width, dst_height, src_stride,
1992
105
                      dst_stride, src, dst, filtering);
1993
105
      return 0;
1994
105
    }
1995
    // 3/8 rounded up for odd sized chroma height.
1996
2.65k
    if (8 * dst_width == 3 * src_width && 8 * dst_height == 3 * src_height) {
1997
      // optimized, 3/8
1998
12
      ScalePlaneDown38(src_width, src_height, dst_width, dst_height, src_stride,
1999
12
                       dst_stride, src, dst, filtering);
2000
12
      return 0;
2001
12
    }
2002
2.64k
    if (4 * dst_width == src_width && 4 * dst_height == src_height &&
2003
52
        (filtering == kFilterBox || filtering == kFilterNone)) {
2004
      // optimized, 1/4
2005
52
      ScalePlaneDown4(src_width, src_height, dst_width, dst_height, src_stride,
2006
52
                      dst_stride, src, dst, filtering);
2007
52
      return 0;
2008
52
    }
2009
2.64k
  }
2010
12.8k
  if (filtering == kFilterBox && dst_height * 2 < src_height) {
2011
1.14k
    return ScalePlaneBox(src_width, src_height, dst_width, dst_height,
2012
1.14k
                         src_stride, dst_stride, src, dst);
2013
1.14k
  }
2014
11.7k
  if ((dst_width + 1) / 2 == src_width && filtering == kFilterLinear) {
2015
290
    ScalePlaneUp2_Linear(src_width, src_height, dst_width, dst_height,
2016
290
                         src_stride, dst_stride, src, dst);
2017
290
    return 0;
2018
290
  }
2019
11.4k
  if ((dst_height + 1) / 2 == src_height && (dst_width + 1) / 2 == src_width &&
2020
291
      (filtering == kFilterBilinear || filtering == kFilterBox)) {
2021
235
    ScalePlaneUp2_Bilinear(src_width, src_height, dst_width, dst_height,
2022
235
                           src_stride, dst_stride, src, dst);
2023
235
    return 0;
2024
235
  }
2025
11.2k
  if (filtering && dst_height > src_height) {
2026
5.42k
    return ScalePlaneBilinearUp(src_width, src_height, dst_width, dst_height,
2027
5.42k
                                src_stride, dst_stride, src, dst, filtering);
2028
5.42k
  }
2029
5.77k
  if (filtering) {
2030
4.48k
    return ScalePlaneBilinearDown(src_width, src_height, dst_width, dst_height,
2031
4.48k
                                  src_stride, dst_stride, src, dst, filtering);
2032
4.48k
  }
2033
1.29k
  ScalePlaneSimple(src_width, src_height, dst_width, dst_height, src_stride,
2034
1.29k
                   dst_stride, src, dst);
2035
1.29k
  return 0;
2036
5.77k
}
2037
2038
LIBYUV_API
2039
int ScalePlane_16(const uint16_t* src,
2040
                  int src_stride,
2041
                  int src_width,
2042
                  int src_height,
2043
                  uint16_t* dst,
2044
                  int dst_stride,
2045
                  int dst_width,
2046
                  int dst_height,
2047
17.7k
                  enum FilterMode filtering) {
2048
  // Reject dimensions larger than 32768 (or smaller than -32768 for height).
2049
  // This prevents FixedDiv signed integer overflows that can lead to division
2050
  // by zero/overflow crashes (SIGFPE on x86) or incorrect step calculations.
2051
17.7k
  if (!src || src_width <= 0 || src_height == 0 || src_width > 32768 ||
2052
17.7k
      src_height < -32768 || src_height > 32768 || !dst || dst_width <= 0 ||
2053
17.7k
      dst_height <= 0) {
2054
0
    return -1;
2055
0
  }
2056
  // Simplify filtering when possible.
2057
17.7k
  filtering = ScaleFilterReduce(src_width, src_height, dst_width, dst_height,
2058
17.7k
                                filtering);
2059
2060
  // Negative height means invert the image.
2061
17.7k
  if (src_height < 0) {
2062
0
    src_height = -src_height;
2063
0
    src = src + (src_height - 1) * (ptrdiff_t)src_stride;
2064
0
    src_stride = -src_stride;
2065
0
  }
2066
  // Use specialized scales to improve performance for common resolutions.
2067
  // For example, all the 1/2 scalings will use ScalePlaneDown2()
2068
17.7k
  if (dst_width == src_width && dst_height == src_height) {
2069
    // Straight copy.
2070
264
    CopyPlane_16(src, src_stride, dst, dst_stride, dst_width, dst_height);
2071
264
    return 0;
2072
264
  }
2073
17.5k
  if (dst_width == src_width && filtering != kFilterBox) {
2074
1.69k
    int dy = 0;
2075
1.69k
    int y = 0;
2076
    // When scaling down, use the center 2 rows to filter.
2077
    // When scaling up, last row of destination uses the last 2 source rows.
2078
1.69k
    if (dst_height <= src_height) {
2079
592
      dy = FixedDiv(src_height, dst_height);
2080
592
      y = CENTERSTART(dy, -32768);  // Subtract 0.5 (32768) to center filter.
2081
      // When scaling up, ensure the last row of destination uses the last
2082
      // source. Avoid divide by zero for dst_height but will do no scaling
2083
      // later.
2084
1.10k
    } else if (src_height > 1 && dst_height > 1) {
2085
987
      dy = FixedDiv1(src_height, dst_height);
2086
987
    }
2087
    // Arbitrary scale vertically, but unscaled horizontally.
2088
1.69k
    ScalePlaneVertical_16(src_height, dst_width, dst_height, src_stride,
2089
1.69k
                          dst_stride, src, dst, 0, y, dy, /*bpp=*/1, filtering);
2090
1.69k
    return 0;
2091
1.69k
  }
2092
15.8k
  if (dst_width <= Abs(src_width) && dst_height <= src_height) {
2093
    // Scale down.
2094
3.71k
    if (4 * dst_width == 3 * src_width && 4 * dst_height == 3 * src_height) {
2095
      // optimized, 3/4
2096
4
      ScalePlaneDown34_16(src_width, src_height, dst_width, dst_height,
2097
4
                          src_stride, dst_stride, src, dst, filtering);
2098
4
      return 0;
2099
4
    }
2100
3.71k
    if (2 * dst_width == src_width && 2 * dst_height == src_height) {
2101
      // optimized, 1/2
2102
93
      ScalePlaneDown2_16(src_width, src_height, dst_width, dst_height,
2103
93
                         src_stride, dst_stride, src, dst, filtering);
2104
93
      return 0;
2105
93
    }
2106
    // 3/8 rounded up for odd sized chroma height.
2107
3.62k
    if (8 * dst_width == 3 * src_width && 8 * dst_height == 3 * src_height) {
2108
      // optimized, 3/8
2109
12
      ScalePlaneDown38_16(src_width, src_height, dst_width, dst_height,
2110
12
                          src_stride, dst_stride, src, dst, filtering);
2111
12
      return 0;
2112
12
    }
2113
3.61k
    if (4 * dst_width == src_width && 4 * dst_height == src_height &&
2114
50
        (filtering == kFilterBox || filtering == kFilterNone)) {
2115
      // optimized, 1/4
2116
50
      ScalePlaneDown4_16(src_width, src_height, dst_width, dst_height,
2117
50
                         src_stride, dst_stride, src, dst, filtering);
2118
50
      return 0;
2119
50
    }
2120
3.61k
  }
2121
15.6k
  if (filtering == kFilterBox && dst_height * 2 < src_height) {
2122
1.75k
    return ScalePlaneBox_16(src_width, src_height, dst_width, dst_height,
2123
1.75k
                            src_stride, dst_stride, src, dst);
2124
1.75k
  }
2125
13.9k
  if ((dst_width + 1) / 2 == src_width && filtering == kFilterLinear) {
2126
0
    ScalePlaneUp2_16_Linear(src_width, src_height, dst_width, dst_height,
2127
0
                            src_stride, dst_stride, src, dst);
2128
0
    return 0;
2129
0
  }
2130
13.9k
  if ((dst_height + 1) / 2 == src_height && (dst_width + 1) / 2 == src_width &&
2131
30
      (filtering == kFilterBilinear || filtering == kFilterBox)) {
2132
0
    ScalePlaneUp2_16_Bilinear(src_width, src_height, dst_width, dst_height,
2133
0
                              src_stride, dst_stride, src, dst);
2134
0
    return 0;
2135
0
  }
2136
13.9k
  if (filtering && dst_height > src_height) {
2137
5.85k
    return ScalePlaneBilinearUp_16(src_width, src_height, dst_width, dst_height,
2138
5.85k
                                   src_stride, dst_stride, src, dst, filtering);
2139
5.85k
  }
2140
8.04k
  if (filtering) {
2141
6.98k
    return ScalePlaneBilinearDown_16(src_width, src_height, dst_width,
2142
6.98k
                                     dst_height, src_stride, dst_stride, src,
2143
6.98k
                                     dst, filtering);
2144
6.98k
  }
2145
1.06k
  ScalePlaneSimple_16(src_width, src_height, dst_width, dst_height, src_stride,
2146
1.06k
                      dst_stride, src, dst);
2147
1.06k
  return 0;
2148
8.04k
}
2149
2150
LIBYUV_API
2151
int ScalePlane_12(const uint16_t* src,
2152
                  int src_stride,
2153
                  int src_width,
2154
                  int src_height,
2155
                  uint16_t* dst,
2156
                  int dst_stride,
2157
                  int dst_width,
2158
                  int dst_height,
2159
18.2k
                  enum FilterMode filtering) {
2160
  // Reject dimensions larger than 32768 (or smaller than -32768 for height).
2161
  // This prevents FixedDiv signed integer overflows that can lead to division
2162
  // by zero/overflow crashes (SIGFPE on x86) or incorrect step calculations.
2163
18.2k
  if (!src || src_width <= 0 || src_height == 0 || src_width > 32768 ||
2164
18.2k
      src_height < -32768 || src_height > 32768 || !dst || dst_width <= 0 ||
2165
18.2k
      dst_height <= 0) {
2166
0
    return -1;
2167
0
  }
2168
  // Simplify filtering when possible.
2169
18.2k
  filtering = ScaleFilterReduce(src_width, src_height, dst_width, dst_height,
2170
18.2k
                                filtering);
2171
2172
  // Negative height means invert the image.
2173
18.2k
  if (src_height < 0) {
2174
0
    src_height = -src_height;
2175
0
    src = src + (src_height - 1) * (ptrdiff_t)src_stride;
2176
0
    src_stride = -src_stride;
2177
0
  }
2178
2179
18.2k
  if ((dst_width + 1) / 2 == src_width && filtering == kFilterLinear) {
2180
283
    ScalePlaneUp2_12_Linear(src_width, src_height, dst_width, dst_height,
2181
283
                            src_stride, dst_stride, src, dst);
2182
283
    return 0;
2183
283
  }
2184
17.9k
  if ((dst_height + 1) / 2 == src_height && (dst_width + 1) / 2 == src_width &&
2185
255
      (filtering == kFilterBilinear || filtering == kFilterBox)) {
2186
178
    ScalePlaneUp2_12_Bilinear(src_width, src_height, dst_width, dst_height,
2187
178
                              src_stride, dst_stride, src, dst);
2188
178
    return 0;
2189
178
  }
2190
2191
17.7k
  return ScalePlane_16(src, src_stride, src_width, src_height, dst, dst_stride,
2192
17.7k
                       dst_width, dst_height, filtering);
2193
17.9k
}
2194
2195
// Scale an I420 image.
2196
// This function in turn calls a scaling function for each plane.
2197
2198
LIBYUV_API
2199
int I420Scale(const uint8_t* src_y,
2200
              int src_stride_y,
2201
              const uint8_t* src_u,
2202
              int src_stride_u,
2203
              const uint8_t* src_v,
2204
              int src_stride_v,
2205
              int src_width,
2206
              int src_height,
2207
              uint8_t* dst_y,
2208
              int dst_stride_y,
2209
              uint8_t* dst_u,
2210
              int dst_stride_u,
2211
              uint8_t* dst_v,
2212
              int dst_stride_v,
2213
              int dst_width,
2214
              int dst_height,
2215
0
              enum FilterMode filtering) {
2216
0
  int r;
2217
2218
0
  if (!src_y || !src_u || !src_v || src_width <= 0 || src_height == 0 ||
2219
0
      src_height == INT_MIN || !dst_y || !dst_u || !dst_v || dst_width <= 0 ||
2220
0
      dst_height <= 0) {
2221
0
    return -1;
2222
0
  }
2223
0
  int src_halfwidth = SUBSAMPLE(src_width, 1, 1);
2224
0
  int src_halfheight = SUBSAMPLE(src_height, 1, 1);
2225
0
  int dst_halfwidth = SUBSAMPLE(dst_width, 1, 1);
2226
0
  int dst_halfheight = SUBSAMPLE(dst_height, 1, 1);
2227
2228
0
  r = ScalePlane(src_y, src_stride_y, src_width, src_height, dst_y,
2229
0
                 dst_stride_y, dst_width, dst_height, filtering);
2230
0
  if (r != 0) {
2231
0
    return r;
2232
0
  }
2233
0
  r = ScalePlane(src_u, src_stride_u, src_halfwidth, src_halfheight, dst_u,
2234
0
                 dst_stride_u, dst_halfwidth, dst_halfheight, filtering);
2235
0
  if (r != 0) {
2236
0
    return r;
2237
0
  }
2238
0
  r = ScalePlane(src_v, src_stride_v, src_halfwidth, src_halfheight, dst_v,
2239
0
                 dst_stride_v, dst_halfwidth, dst_halfheight, filtering);
2240
0
  return r;
2241
0
}
2242
2243
LIBYUV_API
2244
int I420Scale_16(const uint16_t* src_y,
2245
                 int src_stride_y,
2246
                 const uint16_t* src_u,
2247
                 int src_stride_u,
2248
                 const uint16_t* src_v,
2249
                 int src_stride_v,
2250
                 int src_width,
2251
                 int src_height,
2252
                 uint16_t* dst_y,
2253
                 int dst_stride_y,
2254
                 uint16_t* dst_u,
2255
                 int dst_stride_u,
2256
                 uint16_t* dst_v,
2257
                 int dst_stride_v,
2258
                 int dst_width,
2259
                 int dst_height,
2260
0
                 enum FilterMode filtering) {
2261
0
  int r;
2262
2263
0
  if (!src_y || !src_u || !src_v || src_width <= 0 || src_height == 0 ||
2264
0
      src_height == INT_MIN || !dst_y || !dst_u || !dst_v || dst_width <= 0 ||
2265
0
      dst_height <= 0) {
2266
0
    return -1;
2267
0
  }
2268
0
  int src_halfwidth = SUBSAMPLE(src_width, 1, 1);
2269
0
  int src_halfheight = SUBSAMPLE(src_height, 1, 1);
2270
0
  int dst_halfwidth = SUBSAMPLE(dst_width, 1, 1);
2271
0
  int dst_halfheight = SUBSAMPLE(dst_height, 1, 1);
2272
2273
0
  r = ScalePlane_16(src_y, src_stride_y, src_width, src_height, dst_y,
2274
0
                    dst_stride_y, dst_width, dst_height, filtering);
2275
0
  if (r != 0) {
2276
0
    return r;
2277
0
  }
2278
0
  r = ScalePlane_16(src_u, src_stride_u, src_halfwidth, src_halfheight, dst_u,
2279
0
                    dst_stride_u, dst_halfwidth, dst_halfheight, filtering);
2280
0
  if (r != 0) {
2281
0
    return r;
2282
0
  }
2283
0
  r = ScalePlane_16(src_v, src_stride_v, src_halfwidth, src_halfheight, dst_v,
2284
0
                    dst_stride_v, dst_halfwidth, dst_halfheight, filtering);
2285
0
  return r;
2286
0
}
2287
2288
LIBYUV_API
2289
int I420Scale_12(const uint16_t* src_y,
2290
                 int src_stride_y,
2291
                 const uint16_t* src_u,
2292
                 int src_stride_u,
2293
                 const uint16_t* src_v,
2294
                 int src_stride_v,
2295
                 int src_width,
2296
                 int src_height,
2297
                 uint16_t* dst_y,
2298
                 int dst_stride_y,
2299
                 uint16_t* dst_u,
2300
                 int dst_stride_u,
2301
                 uint16_t* dst_v,
2302
                 int dst_stride_v,
2303
                 int dst_width,
2304
                 int dst_height,
2305
0
                 enum FilterMode filtering) {
2306
0
  int r;
2307
2308
0
  if (!src_y || !src_u || !src_v || src_width <= 0 || src_height == 0 ||
2309
0
      src_height == INT_MIN || !dst_y || !dst_u || !dst_v || dst_width <= 0 ||
2310
0
      dst_height <= 0) {
2311
0
    return -1;
2312
0
  }
2313
0
  int src_halfwidth = SUBSAMPLE(src_width, 1, 1);
2314
0
  int src_halfheight = SUBSAMPLE(src_height, 1, 1);
2315
0
  int dst_halfwidth = SUBSAMPLE(dst_width, 1, 1);
2316
0
  int dst_halfheight = SUBSAMPLE(dst_height, 1, 1);
2317
2318
0
  r = ScalePlane_12(src_y, src_stride_y, src_width, src_height, dst_y,
2319
0
                    dst_stride_y, dst_width, dst_height, filtering);
2320
0
  if (r != 0) {
2321
0
    return r;
2322
0
  }
2323
0
  r = ScalePlane_12(src_u, src_stride_u, src_halfwidth, src_halfheight, dst_u,
2324
0
                    dst_stride_u, dst_halfwidth, dst_halfheight, filtering);
2325
0
  if (r != 0) {
2326
0
    return r;
2327
0
  }
2328
0
  r = ScalePlane_12(src_v, src_stride_v, src_halfwidth, src_halfheight, dst_v,
2329
0
                    dst_stride_v, dst_halfwidth, dst_halfheight, filtering);
2330
0
  return r;
2331
0
}
2332
2333
// Scale an I444 image.
2334
// This function in turn calls a scaling function for each plane.
2335
2336
LIBYUV_API
2337
int I444Scale(const uint8_t* src_y,
2338
              int src_stride_y,
2339
              const uint8_t* src_u,
2340
              int src_stride_u,
2341
              const uint8_t* src_v,
2342
              int src_stride_v,
2343
              int src_width,
2344
              int src_height,
2345
              uint8_t* dst_y,
2346
              int dst_stride_y,
2347
              uint8_t* dst_u,
2348
              int dst_stride_u,
2349
              uint8_t* dst_v,
2350
              int dst_stride_v,
2351
              int dst_width,
2352
              int dst_height,
2353
0
              enum FilterMode filtering) {
2354
0
  int r;
2355
2356
0
  if (!src_y || !src_u || !src_v || src_width <= 0 || src_height == 0 ||
2357
0
      src_height == INT_MIN || !dst_y || !dst_u || !dst_v || dst_width <= 0 ||
2358
0
      dst_height <= 0) {
2359
0
    return -1;
2360
0
  }
2361
2362
0
  r = ScalePlane(src_y, src_stride_y, src_width, src_height, dst_y,
2363
0
                 dst_stride_y, dst_width, dst_height, filtering);
2364
0
  if (r != 0) {
2365
0
    return r;
2366
0
  }
2367
0
  r = ScalePlane(src_u, src_stride_u, src_width, src_height, dst_u,
2368
0
                 dst_stride_u, dst_width, dst_height, filtering);
2369
0
  if (r != 0) {
2370
0
    return r;
2371
0
  }
2372
0
  r = ScalePlane(src_v, src_stride_v, src_width, src_height, dst_v,
2373
0
                 dst_stride_v, dst_width, dst_height, filtering);
2374
0
  return r;
2375
0
}
2376
2377
LIBYUV_API
2378
int I444Scale_16(const uint16_t* src_y,
2379
                 int src_stride_y,
2380
                 const uint16_t* src_u,
2381
                 int src_stride_u,
2382
                 const uint16_t* src_v,
2383
                 int src_stride_v,
2384
                 int src_width,
2385
                 int src_height,
2386
                 uint16_t* dst_y,
2387
                 int dst_stride_y,
2388
                 uint16_t* dst_u,
2389
                 int dst_stride_u,
2390
                 uint16_t* dst_v,
2391
                 int dst_stride_v,
2392
                 int dst_width,
2393
                 int dst_height,
2394
0
                 enum FilterMode filtering) {
2395
0
  int r;
2396
2397
0
  if (!src_y || !src_u || !src_v || src_width <= 0 || src_height == 0 ||
2398
0
      src_height == INT_MIN || !dst_y || !dst_u || !dst_v || dst_width <= 0 ||
2399
0
      dst_height <= 0) {
2400
0
    return -1;
2401
0
  }
2402
2403
0
  r = ScalePlane_16(src_y, src_stride_y, src_width, src_height, dst_y,
2404
0
                    dst_stride_y, dst_width, dst_height, filtering);
2405
0
  if (r != 0) {
2406
0
    return r;
2407
0
  }
2408
0
  r = ScalePlane_16(src_u, src_stride_u, src_width, src_height, dst_u,
2409
0
                    dst_stride_u, dst_width, dst_height, filtering);
2410
0
  if (r != 0) {
2411
0
    return r;
2412
0
  }
2413
0
  r = ScalePlane_16(src_v, src_stride_v, src_width, src_height, dst_v,
2414
0
                    dst_stride_v, dst_width, dst_height, filtering);
2415
0
  return r;
2416
0
}
2417
2418
LIBYUV_API
2419
int I444Scale_12(const uint16_t* src_y,
2420
                 int src_stride_y,
2421
                 const uint16_t* src_u,
2422
                 int src_stride_u,
2423
                 const uint16_t* src_v,
2424
                 int src_stride_v,
2425
                 int src_width,
2426
                 int src_height,
2427
                 uint16_t* dst_y,
2428
                 int dst_stride_y,
2429
                 uint16_t* dst_u,
2430
                 int dst_stride_u,
2431
                 uint16_t* dst_v,
2432
                 int dst_stride_v,
2433
                 int dst_width,
2434
                 int dst_height,
2435
0
                 enum FilterMode filtering) {
2436
0
  int r;
2437
2438
0
  if (!src_y || !src_u || !src_v || src_width <= 0 || src_height == 0 ||
2439
0
      src_height == INT_MIN || !dst_y || !dst_u || !dst_v || dst_width <= 0 ||
2440
0
      dst_height <= 0) {
2441
0
    return -1;
2442
0
  }
2443
2444
0
  r = ScalePlane_12(src_y, src_stride_y, src_width, src_height, dst_y,
2445
0
                    dst_stride_y, dst_width, dst_height, filtering);
2446
0
  if (r != 0) {
2447
0
    return r;
2448
0
  }
2449
0
  r = ScalePlane_12(src_u, src_stride_u, src_width, src_height, dst_u,
2450
0
                    dst_stride_u, dst_width, dst_height, filtering);
2451
0
  if (r != 0) {
2452
0
    return r;
2453
0
  }
2454
0
  r = ScalePlane_12(src_v, src_stride_v, src_width, src_height, dst_v,
2455
0
                    dst_stride_v, dst_width, dst_height, filtering);
2456
0
  return r;
2457
0
}
2458
2459
// Scale an I422 image.
2460
// This function in turn calls a scaling function for each plane.
2461
2462
LIBYUV_API
2463
int I422Scale(const uint8_t* src_y,
2464
              int src_stride_y,
2465
              const uint8_t* src_u,
2466
              int src_stride_u,
2467
              const uint8_t* src_v,
2468
              int src_stride_v,
2469
              int src_width,
2470
              int src_height,
2471
              uint8_t* dst_y,
2472
              int dst_stride_y,
2473
              uint8_t* dst_u,
2474
              int dst_stride_u,
2475
              uint8_t* dst_v,
2476
              int dst_stride_v,
2477
              int dst_width,
2478
              int dst_height,
2479
0
              enum FilterMode filtering) {
2480
0
  int r;
2481
2482
0
  if (!src_y || !src_u || !src_v || src_width <= 0 || src_height == 0 ||
2483
0
      src_height == INT_MIN || !dst_y || !dst_u || !dst_v || dst_width <= 0 ||
2484
0
      dst_height <= 0) {
2485
0
    return -1;
2486
0
  }
2487
0
  int src_halfwidth = SUBSAMPLE(src_width, 1, 1);
2488
0
  int dst_halfwidth = SUBSAMPLE(dst_width, 1, 1);
2489
2490
0
  r = ScalePlane(src_y, src_stride_y, src_width, src_height, dst_y,
2491
0
                 dst_stride_y, dst_width, dst_height, filtering);
2492
0
  if (r != 0) {
2493
0
    return r;
2494
0
  }
2495
0
  r = ScalePlane(src_u, src_stride_u, src_halfwidth, src_height, dst_u,
2496
0
                 dst_stride_u, dst_halfwidth, dst_height, filtering);
2497
0
  if (r != 0) {
2498
0
    return r;
2499
0
  }
2500
0
  r = ScalePlane(src_v, src_stride_v, src_halfwidth, src_height, dst_v,
2501
0
                 dst_stride_v, dst_halfwidth, dst_height, filtering);
2502
0
  return r;
2503
0
}
2504
2505
LIBYUV_API
2506
int I422Scale_16(const uint16_t* src_y,
2507
                 int src_stride_y,
2508
                 const uint16_t* src_u,
2509
                 int src_stride_u,
2510
                 const uint16_t* src_v,
2511
                 int src_stride_v,
2512
                 int src_width,
2513
                 int src_height,
2514
                 uint16_t* dst_y,
2515
                 int dst_stride_y,
2516
                 uint16_t* dst_u,
2517
                 int dst_stride_u,
2518
                 uint16_t* dst_v,
2519
                 int dst_stride_v,
2520
                 int dst_width,
2521
                 int dst_height,
2522
0
                 enum FilterMode filtering) {
2523
0
  int r;
2524
2525
0
  if (!src_y || !src_u || !src_v || src_width <= 0 || src_height == 0 ||
2526
0
      src_height == INT_MIN || !dst_y || !dst_u || !dst_v || dst_width <= 0 ||
2527
0
      dst_height <= 0) {
2528
0
    return -1;
2529
0
  }
2530
0
  int src_halfwidth = SUBSAMPLE(src_width, 1, 1);
2531
0
  int dst_halfwidth = SUBSAMPLE(dst_width, 1, 1);
2532
2533
0
  r = ScalePlane_16(src_y, src_stride_y, src_width, src_height, dst_y,
2534
0
                    dst_stride_y, dst_width, dst_height, filtering);
2535
0
  if (r != 0) {
2536
0
    return r;
2537
0
  }
2538
0
  r = ScalePlane_16(src_u, src_stride_u, src_halfwidth, src_height, dst_u,
2539
0
                    dst_stride_u, dst_halfwidth, dst_height, filtering);
2540
0
  if (r != 0) {
2541
0
    return r;
2542
0
  }
2543
0
  r = ScalePlane_16(src_v, src_stride_v, src_halfwidth, src_height, dst_v,
2544
0
                    dst_stride_v, dst_halfwidth, dst_height, filtering);
2545
0
  return r;
2546
0
}
2547
2548
LIBYUV_API
2549
int I422Scale_12(const uint16_t* src_y,
2550
                 int src_stride_y,
2551
                 const uint16_t* src_u,
2552
                 int src_stride_u,
2553
                 const uint16_t* src_v,
2554
                 int src_stride_v,
2555
                 int src_width,
2556
                 int src_height,
2557
                 uint16_t* dst_y,
2558
                 int dst_stride_y,
2559
                 uint16_t* dst_u,
2560
                 int dst_stride_u,
2561
                 uint16_t* dst_v,
2562
                 int dst_stride_v,
2563
                 int dst_width,
2564
                 int dst_height,
2565
0
                 enum FilterMode filtering) {
2566
0
  int r;
2567
2568
0
  if (!src_y || !src_u || !src_v || src_width <= 0 || src_height == 0 ||
2569
0
      src_height == INT_MIN || !dst_y || !dst_u || !dst_v || dst_width <= 0 ||
2570
0
      dst_height <= 0) {
2571
0
    return -1;
2572
0
  }
2573
0
  int src_halfwidth = SUBSAMPLE(src_width, 1, 1);
2574
0
  int dst_halfwidth = SUBSAMPLE(dst_width, 1, 1);
2575
2576
0
  r = ScalePlane_12(src_y, src_stride_y, src_width, src_height, dst_y,
2577
0
                    dst_stride_y, dst_width, dst_height, filtering);
2578
0
  if (r != 0) {
2579
0
    return r;
2580
0
  }
2581
0
  r = ScalePlane_12(src_u, src_stride_u, src_halfwidth, src_height, dst_u,
2582
0
                    dst_stride_u, dst_halfwidth, dst_height, filtering);
2583
0
  if (r != 0) {
2584
0
    return r;
2585
0
  }
2586
0
  r = ScalePlane_12(src_v, src_stride_v, src_halfwidth, src_height, dst_v,
2587
0
                    dst_stride_v, dst_halfwidth, dst_height, filtering);
2588
0
  return r;
2589
0
}
2590
2591
// Scale an NV12 image.
2592
// This function in turn calls a scaling function for each plane.
2593
2594
LIBYUV_API
2595
int NV12Scale(const uint8_t* src_y,
2596
              int src_stride_y,
2597
              const uint8_t* src_uv,
2598
              int src_stride_uv,
2599
              int src_width,
2600
              int src_height,
2601
              uint8_t* dst_y,
2602
              int dst_stride_y,
2603
              uint8_t* dst_uv,
2604
              int dst_stride_uv,
2605
              int dst_width,
2606
              int dst_height,
2607
0
              enum FilterMode filtering) {
2608
0
  int r;
2609
2610
0
  if (!src_y || !src_uv || src_width <= 0 || src_height == 0 ||
2611
0
      src_height == INT_MIN || !dst_y || !dst_uv || dst_width <= 0 ||
2612
0
      dst_height <= 0) {
2613
0
    return -1;
2614
0
  }
2615
0
  int src_halfwidth = SUBSAMPLE(src_width, 1, 1);
2616
0
  int src_halfheight = SUBSAMPLE(src_height, 1, 1);
2617
0
  int dst_halfwidth = SUBSAMPLE(dst_width, 1, 1);
2618
0
  int dst_halfheight = SUBSAMPLE(dst_height, 1, 1);
2619
2620
0
  r = ScalePlane(src_y, src_stride_y, src_width, src_height, dst_y,
2621
0
                 dst_stride_y, dst_width, dst_height, filtering);
2622
0
  if (r != 0) {
2623
0
    return r;
2624
0
  }
2625
0
  r = UVScale(src_uv, src_stride_uv, src_halfwidth, src_halfheight, dst_uv,
2626
0
              dst_stride_uv, dst_halfwidth, dst_halfheight, filtering);
2627
0
  return r;
2628
0
}
2629
2630
LIBYUV_API
2631
int NV24Scale(const uint8_t* src_y,
2632
              int src_stride_y,
2633
              const uint8_t* src_uv,
2634
              int src_stride_uv,
2635
              int src_width,
2636
              int src_height,
2637
              uint8_t* dst_y,
2638
              int dst_stride_y,
2639
              uint8_t* dst_uv,
2640
              int dst_stride_uv,
2641
              int dst_width,
2642
              int dst_height,
2643
0
              enum FilterMode filtering) {
2644
0
  int r;
2645
2646
0
  if (!src_y || !src_uv || src_width <= 0 || src_height == 0 ||
2647
0
      src_height == INT_MIN || !dst_y || !dst_uv || dst_width <= 0 ||
2648
0
      dst_height <= 0) {
2649
0
    return -1;
2650
0
  }
2651
2652
0
  r = ScalePlane(src_y, src_stride_y, src_width, src_height, dst_y,
2653
0
                 dst_stride_y, dst_width, dst_height, filtering);
2654
0
  if (r != 0) {
2655
0
    return r;
2656
0
  }
2657
0
  r = UVScale(src_uv, src_stride_uv, src_width, src_height, dst_uv,
2658
0
              dst_stride_uv, dst_width, dst_height, filtering);
2659
0
  return r;
2660
0
}
2661
2662
// Deprecated api
2663
LIBYUV_API
2664
int Scale(const uint8_t* src_y,
2665
          const uint8_t* src_u,
2666
          const uint8_t* src_v,
2667
          int src_stride_y,
2668
          int src_stride_u,
2669
          int src_stride_v,
2670
          int src_width,
2671
          int src_height,
2672
          uint8_t* dst_y,
2673
          uint8_t* dst_u,
2674
          uint8_t* dst_v,
2675
          int dst_stride_y,
2676
          int dst_stride_u,
2677
          int dst_stride_v,
2678
          int dst_width,
2679
          int dst_height,
2680
0
          LIBYUV_BOOL interpolate) {
2681
0
  return I420Scale(src_y, src_stride_y, src_u, src_stride_u, src_v,
2682
0
                   src_stride_v, src_width, src_height, dst_y, dst_stride_y,
2683
0
                   dst_u, dst_stride_u, dst_v, dst_stride_v, dst_width,
2684
0
                   dst_height, interpolate ? kFilterBox : kFilterNone);
2685
0
}
2686
2687
#ifdef __cplusplus
2688
}  // extern "C"
2689
}  // namespace libyuv
2690
#endif