/src/libavif/ext/libyuv/source/scale.cc
Line | Count | Source |
1 | | /* |
2 | | * Copyright 2011 The LibYuv Project Authors. All rights reserved. |
3 | | * |
4 | | * Use of this source code is governed by a BSD-style license |
5 | | * that can be found in the LICENSE file in the root of the source |
6 | | * tree. An additional intellectual property rights grant can be found |
7 | | * in the file PATENTS. All contributing project authors may |
8 | | * be found in the AUTHORS file in the root of the source tree. |
9 | | */ |
10 | | |
11 | | #include "libyuv/scale.h" |
12 | | |
13 | | #include <assert.h> |
14 | | #include <limits.h> |
15 | | #include <string.h> |
16 | | |
17 | | #include "libyuv/cpu_id.h" |
18 | | #include "libyuv/planar_functions.h" // For CopyPlane |
19 | | #include "libyuv/row.h" |
20 | | #include "libyuv/scale_row.h" |
21 | | #include "libyuv/scale_uv.h" // For UVScale |
22 | | |
23 | | #ifdef __cplusplus |
24 | | namespace libyuv { |
25 | | extern "C" { |
26 | | #endif |
27 | | |
28 | 56.8k | static __inline int Abs(int v) { |
29 | 56.8k | return v >= 0 ? v : -v; |
30 | 56.8k | } |
31 | | |
32 | 0 | #define SUBSAMPLE(v, a, s) (v < 0) ? (-((-v + a) >> s)) : ((v + a) >> s) |
33 | 1.32k | #define CENTERSTART(dx, s) (dx < 0) ? -((-dx >> 1) + s) : ((dx >> 1) + s) |
34 | | |
35 | | // Scale plane, 1/2 |
36 | | // This is an optimized version for scaling down a plane to 1/2 of |
37 | | // its original size. |
38 | | |
39 | | static void ScalePlaneDown2(int src_width, |
40 | | int src_height, |
41 | | int dst_width, |
42 | | int dst_height, |
43 | | ptrdiff_t src_stride, |
44 | | ptrdiff_t dst_stride, |
45 | | const uint8_t* src_ptr, |
46 | | uint8_t* dst_ptr, |
47 | 105 | enum FilterMode filtering) { |
48 | 105 | int y; |
49 | 105 | void (*ScaleRowDown2)(const uint8_t* src_ptr, ptrdiff_t src_stride, |
50 | 105 | uint8_t* dst_ptr, int dst_width) = |
51 | 105 | filtering == kFilterNone |
52 | 105 | ? ScaleRowDown2_C |
53 | 105 | : (filtering == kFilterLinear ? ScaleRowDown2Linear_C |
54 | 105 | : ScaleRowDown2Box_C); |
55 | 105 | ptrdiff_t row_stride = src_stride * 2; |
56 | 105 | (void)src_width; |
57 | 105 | (void)src_height; |
58 | 105 | if (!filtering) { |
59 | 0 | src_ptr += src_stride; // Point to odd rows. |
60 | 0 | src_stride = 0; |
61 | 0 | } |
62 | | |
63 | | #if defined(HAS_SCALEROWDOWN2_NEON) |
64 | | if (TestCpuFlag(kCpuHasNEON)) { |
65 | | ScaleRowDown2 = |
66 | | filtering == kFilterNone |
67 | | ? ScaleRowDown2_Any_NEON |
68 | | : (filtering == kFilterLinear ? ScaleRowDown2Linear_Any_NEON |
69 | | : ScaleRowDown2Box_Any_NEON); |
70 | | if (IS_ALIGNED(dst_width, 16)) { |
71 | | ScaleRowDown2 = filtering == kFilterNone ? ScaleRowDown2_NEON |
72 | | : (filtering == kFilterLinear |
73 | | ? ScaleRowDown2Linear_NEON |
74 | | : ScaleRowDown2Box_NEON); |
75 | | } |
76 | | } |
77 | | #endif |
78 | | #if defined(HAS_SCALEROWDOWN2_SME) |
79 | | if (TestCpuFlag(kCpuHasSME)) { |
80 | | ScaleRowDown2 = filtering == kFilterNone ? ScaleRowDown2_SME |
81 | | : filtering == kFilterLinear ? ScaleRowDown2Linear_SME |
82 | | : ScaleRowDown2Box_SME; |
83 | | } |
84 | | #endif |
85 | 105 | #if defined(HAS_SCALEROWDOWN2_SSSE3) |
86 | 105 | if (TestCpuFlag(kCpuHasSSSE3)) { |
87 | 105 | ScaleRowDown2 = |
88 | 105 | filtering == kFilterNone |
89 | 105 | ? ScaleRowDown2_Any_SSSE3 |
90 | 105 | : (filtering == kFilterLinear ? ScaleRowDown2Linear_Any_SSSE3 |
91 | 105 | : ScaleRowDown2Box_Any_SSSE3); |
92 | 105 | if (IS_ALIGNED(dst_width, 16)) { |
93 | 12 | ScaleRowDown2 = |
94 | 12 | filtering == kFilterNone |
95 | 12 | ? ScaleRowDown2_SSSE3 |
96 | 12 | : (filtering == kFilterLinear ? ScaleRowDown2Linear_SSSE3 |
97 | 12 | : ScaleRowDown2Box_SSSE3); |
98 | 12 | } |
99 | 105 | } |
100 | 105 | #endif |
101 | 105 | #if defined(HAS_SCALEROWDOWN2_AVX2) |
102 | 105 | if (TestCpuFlag(kCpuHasAVX2)) { |
103 | 105 | ScaleRowDown2 = |
104 | 105 | filtering == kFilterNone |
105 | 105 | ? ScaleRowDown2_Any_AVX2 |
106 | 105 | : (filtering == kFilterLinear ? ScaleRowDown2Linear_Any_AVX2 |
107 | 105 | : ScaleRowDown2Box_Any_AVX2); |
108 | 105 | if (IS_ALIGNED(dst_width, 32)) { |
109 | 12 | ScaleRowDown2 = filtering == kFilterNone ? ScaleRowDown2_AVX2 |
110 | 12 | : (filtering == kFilterLinear |
111 | 12 | ? ScaleRowDown2Linear_AVX2 |
112 | 12 | : ScaleRowDown2Box_AVX2); |
113 | 12 | } |
114 | 105 | } |
115 | 105 | #endif |
116 | | #if defined(HAS_SCALEROWDOWN2_LSX) |
117 | | if (TestCpuFlag(kCpuHasLSX)) { |
118 | | ScaleRowDown2 = |
119 | | filtering == kFilterNone |
120 | | ? ScaleRowDown2_Any_LSX |
121 | | : (filtering == kFilterLinear ? ScaleRowDown2Linear_Any_LSX |
122 | | : ScaleRowDown2Box_Any_LSX); |
123 | | if (IS_ALIGNED(dst_width, 32)) { |
124 | | ScaleRowDown2 = filtering == kFilterNone ? ScaleRowDown2_LSX |
125 | | : (filtering == kFilterLinear |
126 | | ? ScaleRowDown2Linear_LSX |
127 | | : ScaleRowDown2Box_LSX); |
128 | | } |
129 | | } |
130 | | #endif |
131 | | #if defined(HAS_SCALEROWDOWN2_RVV) |
132 | | if (TestCpuFlag(kCpuHasRVV)) { |
133 | | ScaleRowDown2 = filtering == kFilterNone |
134 | | ? ScaleRowDown2_RVV |
135 | | : (filtering == kFilterLinear ? ScaleRowDown2Linear_RVV |
136 | | : ScaleRowDown2Box_RVV); |
137 | | } |
138 | | #endif |
139 | | |
140 | 105 | if (filtering == kFilterLinear) { |
141 | 0 | src_stride = 0; |
142 | 0 | } |
143 | | // TODO(fbarchard): Loop through source height to allow odd height. |
144 | 960 | for (y = 0; y < dst_height; ++y) { |
145 | 855 | ScaleRowDown2(src_ptr, src_stride, dst_ptr, dst_width); |
146 | 855 | src_ptr += row_stride; |
147 | 855 | dst_ptr += dst_stride; |
148 | 855 | } |
149 | 105 | } |
150 | | |
151 | | static void ScalePlaneDown2_16(int src_width, |
152 | | int src_height, |
153 | | int dst_width, |
154 | | int dst_height, |
155 | | ptrdiff_t src_stride, |
156 | | ptrdiff_t dst_stride, |
157 | | const uint16_t* src_ptr, |
158 | | uint16_t* dst_ptr, |
159 | 93 | enum FilterMode filtering) { |
160 | 93 | int y; |
161 | 93 | void (*ScaleRowDown2)(const uint16_t* src_ptr, ptrdiff_t src_stride, |
162 | 93 | uint16_t* dst_ptr, int dst_width) = |
163 | 93 | filtering == kFilterNone |
164 | 93 | ? ScaleRowDown2_16_C |
165 | 93 | : (filtering == kFilterLinear ? ScaleRowDown2Linear_16_C |
166 | 93 | : ScaleRowDown2Box_16_C); |
167 | 93 | ptrdiff_t row_stride = src_stride * 2; |
168 | 93 | (void)src_width; |
169 | 93 | (void)src_height; |
170 | 93 | if (!filtering) { |
171 | 0 | src_ptr += src_stride; // Point to odd rows. |
172 | 0 | src_stride = 0; |
173 | 0 | } |
174 | | |
175 | | #if defined(HAS_SCALEROWDOWN2_16_NEON) |
176 | | if (TestCpuFlag(kCpuHasNEON) && IS_ALIGNED(dst_width, 16)) { |
177 | | ScaleRowDown2 = filtering == kFilterNone ? ScaleRowDown2_16_NEON |
178 | | : filtering == kFilterLinear ? ScaleRowDown2Linear_16_NEON |
179 | | : ScaleRowDown2Box_16_NEON; |
180 | | } |
181 | | #endif |
182 | | #if defined(HAS_SCALEROWDOWN2_16_SME) |
183 | | if (TestCpuFlag(kCpuHasSME)) { |
184 | | ScaleRowDown2 = filtering == kFilterNone ? ScaleRowDown2_16_SME |
185 | | : filtering == kFilterLinear ? ScaleRowDown2Linear_16_SME |
186 | | : ScaleRowDown2Box_16_SME; |
187 | | } |
188 | | #endif |
189 | | #if defined(HAS_SCALEROWDOWN2_16_SSE2) |
190 | | if (TestCpuFlag(kCpuHasSSE2) && IS_ALIGNED(dst_width, 16)) { |
191 | | ScaleRowDown2 = |
192 | | filtering == kFilterNone |
193 | | ? ScaleRowDown2_16_SSE2 |
194 | | : (filtering == kFilterLinear ? ScaleRowDown2Linear_16_SSE2 |
195 | | : ScaleRowDown2Box_16_SSE2); |
196 | | } |
197 | | #endif |
198 | | |
199 | 93 | if (filtering == kFilterLinear) { |
200 | 0 | src_stride = 0; |
201 | 0 | } |
202 | | // TODO(fbarchard): Loop through source height to allow odd height. |
203 | 837 | for (y = 0; y < dst_height; ++y) { |
204 | 744 | ScaleRowDown2(src_ptr, src_stride, dst_ptr, dst_width); |
205 | 744 | src_ptr += row_stride; |
206 | 744 | dst_ptr += dst_stride; |
207 | 744 | } |
208 | 93 | } |
209 | | |
210 | | |
211 | | // Scale plane, 1/4 |
212 | | // This is an optimized version for scaling down a plane to 1/4 of |
213 | | // its original size. |
214 | | |
215 | | static void ScalePlaneDown4(int src_width, |
216 | | int src_height, |
217 | | int dst_width, |
218 | | int dst_height, |
219 | | ptrdiff_t src_stride, |
220 | | ptrdiff_t dst_stride, |
221 | | const uint8_t* src_ptr, |
222 | | uint8_t* dst_ptr, |
223 | 52 | enum FilterMode filtering) { |
224 | 52 | int y; |
225 | 52 | void (*ScaleRowDown4)(const uint8_t* src_ptr, ptrdiff_t src_stride, |
226 | 52 | uint8_t* dst_ptr, int dst_width) = |
227 | 52 | filtering ? ScaleRowDown4Box_C : ScaleRowDown4_C; |
228 | 52 | ptrdiff_t row_stride = src_stride * 4; |
229 | 52 | (void)src_width; |
230 | 52 | (void)src_height; |
231 | 52 | if (!filtering) { |
232 | 0 | src_ptr += src_stride * 2; // Point to row 2. |
233 | 0 | src_stride = 0; |
234 | 0 | } |
235 | | #if defined(HAS_SCALEROWDOWN4_NEON) |
236 | | if (TestCpuFlag(kCpuHasNEON)) { |
237 | | ScaleRowDown4 = |
238 | | filtering ? ScaleRowDown4Box_Any_NEON : ScaleRowDown4_Any_NEON; |
239 | | if (IS_ALIGNED(dst_width, 16)) { |
240 | | ScaleRowDown4 = filtering ? ScaleRowDown4Box_NEON : ScaleRowDown4_NEON; |
241 | | } |
242 | | } |
243 | | #endif |
244 | 52 | #if defined(HAS_SCALEROWDOWN4_SSSE3) |
245 | 52 | if (TestCpuFlag(kCpuHasSSSE3)) { |
246 | 52 | ScaleRowDown4 = |
247 | 52 | filtering ? ScaleRowDown4Box_Any_SSSE3 : ScaleRowDown4_Any_SSSE3; |
248 | 52 | if (IS_ALIGNED(dst_width, 8)) { |
249 | 0 | ScaleRowDown4 = filtering ? ScaleRowDown4Box_SSSE3 : ScaleRowDown4_SSSE3; |
250 | 0 | } |
251 | 52 | } |
252 | 52 | #endif |
253 | 52 | #if defined(HAS_SCALEROWDOWN4_AVX2) |
254 | 52 | if (TestCpuFlag(kCpuHasAVX2)) { |
255 | 52 | ScaleRowDown4 = |
256 | 52 | filtering ? ScaleRowDown4Box_Any_AVX2 : ScaleRowDown4_Any_AVX2; |
257 | 52 | if (IS_ALIGNED(dst_width, 16)) { |
258 | 0 | ScaleRowDown4 = filtering ? ScaleRowDown4Box_AVX2 : ScaleRowDown4_AVX2; |
259 | 0 | } |
260 | 52 | } |
261 | 52 | #endif |
262 | | #if defined(HAS_SCALEROWDOWN4_LSX) |
263 | | if (TestCpuFlag(kCpuHasLSX)) { |
264 | | ScaleRowDown4 = |
265 | | filtering ? ScaleRowDown4Box_Any_LSX : ScaleRowDown4_Any_LSX; |
266 | | if (IS_ALIGNED(dst_width, 16)) { |
267 | | ScaleRowDown4 = filtering ? ScaleRowDown4Box_LSX : ScaleRowDown4_LSX; |
268 | | } |
269 | | } |
270 | | #endif |
271 | | #if defined(HAS_SCALEROWDOWN4_RVV) |
272 | | if (TestCpuFlag(kCpuHasRVV)) { |
273 | | ScaleRowDown4 = filtering ? ScaleRowDown4Box_RVV : ScaleRowDown4_RVV; |
274 | | } |
275 | | #endif |
276 | | |
277 | 52 | if (filtering == kFilterLinear) { |
278 | 0 | src_stride = 0; |
279 | 0 | } |
280 | 547 | for (y = 0; y < dst_height; ++y) { |
281 | 495 | ScaleRowDown4(src_ptr, src_stride, dst_ptr, dst_width); |
282 | 495 | src_ptr += row_stride; |
283 | 495 | dst_ptr += dst_stride; |
284 | 495 | } |
285 | 52 | } |
286 | | |
287 | | static void ScalePlaneDown4_16(int src_width, |
288 | | int src_height, |
289 | | int dst_width, |
290 | | int dst_height, |
291 | | ptrdiff_t src_stride, |
292 | | ptrdiff_t dst_stride, |
293 | | const uint16_t* src_ptr, |
294 | | uint16_t* dst_ptr, |
295 | 50 | enum FilterMode filtering) { |
296 | 50 | int y; |
297 | 50 | void (*ScaleRowDown4)(const uint16_t* src_ptr, ptrdiff_t src_stride, |
298 | 50 | uint16_t* dst_ptr, int dst_width) = |
299 | 50 | filtering ? ScaleRowDown4Box_16_C : ScaleRowDown4_16_C; |
300 | 50 | ptrdiff_t row_stride = src_stride * 4; |
301 | 50 | (void)src_width; |
302 | 50 | (void)src_height; |
303 | 50 | if (!filtering) { |
304 | 0 | src_ptr += src_stride * 2; // Point to row 2. |
305 | 0 | src_stride = 0; |
306 | 0 | } |
307 | | #if defined(HAS_SCALEROWDOWN4_16_NEON) |
308 | | if (TestCpuFlag(kCpuHasNEON) && IS_ALIGNED(dst_width, 8)) { |
309 | | ScaleRowDown4 = |
310 | | filtering ? ScaleRowDown4Box_16_NEON : ScaleRowDown4_16_NEON; |
311 | | } |
312 | | #endif |
313 | | #if defined(HAS_SCALEROWDOWN4_16_SSE2) |
314 | | if (TestCpuFlag(kCpuHasSSE2) && IS_ALIGNED(dst_width, 8)) { |
315 | | ScaleRowDown4 = |
316 | | filtering ? ScaleRowDown4Box_16_SSE2 : ScaleRowDown4_16_SSE2; |
317 | | } |
318 | | #endif |
319 | | |
320 | 50 | if (filtering == kFilterLinear) { |
321 | 0 | src_stride = 0; |
322 | 0 | } |
323 | 579 | for (y = 0; y < dst_height; ++y) { |
324 | 529 | ScaleRowDown4(src_ptr, src_stride, dst_ptr, dst_width); |
325 | 529 | src_ptr += row_stride; |
326 | 529 | dst_ptr += dst_stride; |
327 | 529 | } |
328 | 50 | } |
329 | | |
330 | | // Scale plane down, 3/4 |
331 | | static void ScalePlaneDown34(int src_width, |
332 | | int src_height, |
333 | | int dst_width, |
334 | | int dst_height, |
335 | | ptrdiff_t src_stride, |
336 | | ptrdiff_t dst_stride, |
337 | | const uint8_t* src_ptr, |
338 | | uint8_t* dst_ptr, |
339 | 2 | enum FilterMode filtering) { |
340 | 2 | int y; |
341 | 2 | void (*ScaleRowDown34_0)(const uint8_t* src_ptr, ptrdiff_t src_stride, |
342 | 2 | uint8_t* dst_ptr, int dst_width); |
343 | 2 | void (*ScaleRowDown34_1)(const uint8_t* src_ptr, ptrdiff_t src_stride, |
344 | 2 | uint8_t* dst_ptr, int dst_width); |
345 | 2 | const ptrdiff_t filter_stride = (filtering == kFilterLinear) ? 0 : src_stride; |
346 | 2 | (void)src_width; |
347 | 2 | (void)src_height; |
348 | 2 | assert(dst_width % 3 == 0); |
349 | 2 | if (!filtering) { |
350 | 0 | ScaleRowDown34_0 = ScaleRowDown34_C; |
351 | 0 | ScaleRowDown34_1 = ScaleRowDown34_C; |
352 | 2 | } else { |
353 | 2 | ScaleRowDown34_0 = ScaleRowDown34_0_Box_C; |
354 | 2 | ScaleRowDown34_1 = ScaleRowDown34_1_Box_C; |
355 | 2 | } |
356 | | #if defined(HAS_SCALEROWDOWN34_NEON) |
357 | | if (TestCpuFlag(kCpuHasNEON)) { |
358 | | #if defined(__aarch64__) |
359 | | if (dst_width % 48 == 0) { |
360 | | #else |
361 | | if (dst_width % 24 == 0) { |
362 | | #endif |
363 | | if (!filtering) { |
364 | | ScaleRowDown34_0 = ScaleRowDown34_NEON; |
365 | | ScaleRowDown34_1 = ScaleRowDown34_NEON; |
366 | | } else { |
367 | | ScaleRowDown34_0 = ScaleRowDown34_0_Box_NEON; |
368 | | ScaleRowDown34_1 = ScaleRowDown34_1_Box_NEON; |
369 | | } |
370 | | } else { |
371 | | if (!filtering) { |
372 | | ScaleRowDown34_0 = ScaleRowDown34_Any_NEON; |
373 | | ScaleRowDown34_1 = ScaleRowDown34_Any_NEON; |
374 | | } else { |
375 | | ScaleRowDown34_0 = ScaleRowDown34_0_Box_Any_NEON; |
376 | | ScaleRowDown34_1 = ScaleRowDown34_1_Box_Any_NEON; |
377 | | } |
378 | | } |
379 | | } |
380 | | #endif |
381 | | #if defined(HAS_SCALEROWDOWN34_LSX) |
382 | | if (TestCpuFlag(kCpuHasLSX)) { |
383 | | if (dst_width % 48 == 0) { |
384 | | if (!filtering) { |
385 | | ScaleRowDown34_0 = ScaleRowDown34_LSX; |
386 | | ScaleRowDown34_1 = ScaleRowDown34_LSX; |
387 | | } else { |
388 | | ScaleRowDown34_0 = ScaleRowDown34_0_Box_LSX; |
389 | | ScaleRowDown34_1 = ScaleRowDown34_1_Box_LSX; |
390 | | } |
391 | | } else { |
392 | | if (!filtering) { |
393 | | ScaleRowDown34_0 = ScaleRowDown34_Any_LSX; |
394 | | ScaleRowDown34_1 = ScaleRowDown34_Any_LSX; |
395 | | } else { |
396 | | ScaleRowDown34_0 = ScaleRowDown34_0_Box_Any_LSX; |
397 | | ScaleRowDown34_1 = ScaleRowDown34_1_Box_Any_LSX; |
398 | | } |
399 | | } |
400 | | } |
401 | | #endif |
402 | 2 | #if defined(HAS_SCALEROWDOWN34_SSSE3) |
403 | 2 | if (TestCpuFlag(kCpuHasSSSE3)) { |
404 | 2 | if (dst_width % 24 == 0) { |
405 | 0 | if (!filtering) { |
406 | 0 | ScaleRowDown34_0 = ScaleRowDown34_SSSE3; |
407 | 0 | ScaleRowDown34_1 = ScaleRowDown34_SSSE3; |
408 | 0 | } else { |
409 | 0 | ScaleRowDown34_0 = ScaleRowDown34_0_Box_SSSE3; |
410 | 0 | ScaleRowDown34_1 = ScaleRowDown34_1_Box_SSSE3; |
411 | 0 | } |
412 | 2 | } else { |
413 | 2 | if (!filtering) { |
414 | 0 | ScaleRowDown34_0 = ScaleRowDown34_Any_SSSE3; |
415 | 0 | ScaleRowDown34_1 = ScaleRowDown34_Any_SSSE3; |
416 | 2 | } else { |
417 | 2 | ScaleRowDown34_0 = ScaleRowDown34_0_Box_Any_SSSE3; |
418 | 2 | ScaleRowDown34_1 = ScaleRowDown34_1_Box_Any_SSSE3; |
419 | 2 | } |
420 | 2 | } |
421 | 2 | } |
422 | 2 | #endif |
423 | | #if defined(HAS_SCALEROWDOWN34_RVV) |
424 | | if (TestCpuFlag(kCpuHasRVV)) { |
425 | | if (!filtering) { |
426 | | ScaleRowDown34_0 = ScaleRowDown34_RVV; |
427 | | ScaleRowDown34_1 = ScaleRowDown34_RVV; |
428 | | } else { |
429 | | ScaleRowDown34_0 = ScaleRowDown34_0_Box_RVV; |
430 | | ScaleRowDown34_1 = ScaleRowDown34_1_Box_RVV; |
431 | | } |
432 | | } |
433 | | #endif |
434 | | |
435 | 4 | for (y = 0; y < dst_height - 2; y += 3) { |
436 | 2 | ScaleRowDown34_0(src_ptr, filter_stride, dst_ptr, dst_width); |
437 | 2 | src_ptr += src_stride; |
438 | 2 | dst_ptr += dst_stride; |
439 | 2 | ScaleRowDown34_1(src_ptr, filter_stride, dst_ptr, dst_width); |
440 | 2 | src_ptr += src_stride; |
441 | 2 | dst_ptr += dst_stride; |
442 | 2 | ScaleRowDown34_0(src_ptr + src_stride, -filter_stride, dst_ptr, dst_width); |
443 | 2 | src_ptr += src_stride * 2; |
444 | 2 | dst_ptr += dst_stride; |
445 | 2 | } |
446 | | |
447 | | // Remainder 1 or 2 rows with last row vertically unfiltered |
448 | 2 | if ((dst_height % 3) == 2) { |
449 | 0 | ScaleRowDown34_0(src_ptr, filter_stride, dst_ptr, dst_width); |
450 | 0 | src_ptr += src_stride; |
451 | 0 | dst_ptr += dst_stride; |
452 | 0 | ScaleRowDown34_1(src_ptr, 0, dst_ptr, dst_width); |
453 | 2 | } else if ((dst_height % 3) == 1) { |
454 | 0 | ScaleRowDown34_0(src_ptr, 0, dst_ptr, dst_width); |
455 | 0 | } |
456 | 2 | } |
457 | | |
458 | | static void ScalePlaneDown34_16(int src_width, |
459 | | int src_height, |
460 | | int dst_width, |
461 | | int dst_height, |
462 | | ptrdiff_t src_stride, |
463 | | ptrdiff_t dst_stride, |
464 | | const uint16_t* src_ptr, |
465 | | uint16_t* dst_ptr, |
466 | 4 | enum FilterMode filtering) { |
467 | 4 | int y; |
468 | 4 | void (*ScaleRowDown34_0)(const uint16_t* src_ptr, ptrdiff_t src_stride, |
469 | 4 | uint16_t* dst_ptr, int dst_width); |
470 | 4 | void (*ScaleRowDown34_1)(const uint16_t* src_ptr, ptrdiff_t src_stride, |
471 | 4 | uint16_t* dst_ptr, int dst_width); |
472 | 4 | const ptrdiff_t filter_stride = (filtering == kFilterLinear) ? 0 : src_stride; |
473 | 4 | (void)src_width; |
474 | 4 | (void)src_height; |
475 | 4 | assert(dst_width % 3 == 0); |
476 | 4 | if (!filtering) { |
477 | 0 | ScaleRowDown34_0 = ScaleRowDown34_16_C; |
478 | 0 | ScaleRowDown34_1 = ScaleRowDown34_16_C; |
479 | 4 | } else { |
480 | 4 | ScaleRowDown34_0 = ScaleRowDown34_0_Box_16_C; |
481 | 4 | ScaleRowDown34_1 = ScaleRowDown34_1_Box_16_C; |
482 | 4 | } |
483 | | #if defined(HAS_SCALEROWDOWN34_16_NEON) |
484 | | if (TestCpuFlag(kCpuHasNEON) && (dst_width % 24 == 0)) { |
485 | | if (!filtering) { |
486 | | ScaleRowDown34_0 = ScaleRowDown34_16_NEON; |
487 | | ScaleRowDown34_1 = ScaleRowDown34_16_NEON; |
488 | | } else { |
489 | | ScaleRowDown34_0 = ScaleRowDown34_0_Box_16_NEON; |
490 | | ScaleRowDown34_1 = ScaleRowDown34_1_Box_16_NEON; |
491 | | } |
492 | | } |
493 | | #endif |
494 | | #if defined(HAS_SCALEROWDOWN34_16_SSSE3) |
495 | | if (TestCpuFlag(kCpuHasSSSE3) && (dst_width % 24 == 0)) { |
496 | | if (!filtering) { |
497 | | ScaleRowDown34_0 = ScaleRowDown34_16_SSSE3; |
498 | | ScaleRowDown34_1 = ScaleRowDown34_16_SSSE3; |
499 | | } else { |
500 | | ScaleRowDown34_0 = ScaleRowDown34_0_Box_16_SSSE3; |
501 | | ScaleRowDown34_1 = ScaleRowDown34_1_Box_16_SSSE3; |
502 | | } |
503 | | } |
504 | | #endif |
505 | | |
506 | 28 | for (y = 0; y < dst_height - 2; y += 3) { |
507 | 24 | ScaleRowDown34_0(src_ptr, filter_stride, dst_ptr, dst_width); |
508 | 24 | src_ptr += src_stride; |
509 | 24 | dst_ptr += dst_stride; |
510 | 24 | ScaleRowDown34_1(src_ptr, filter_stride, dst_ptr, dst_width); |
511 | 24 | src_ptr += src_stride; |
512 | 24 | dst_ptr += dst_stride; |
513 | 24 | ScaleRowDown34_0(src_ptr + src_stride, -filter_stride, dst_ptr, dst_width); |
514 | 24 | src_ptr += src_stride * 2; |
515 | 24 | dst_ptr += dst_stride; |
516 | 24 | } |
517 | | |
518 | | // Remainder 1 or 2 rows with last row vertically unfiltered |
519 | 4 | if ((dst_height % 3) == 2) { |
520 | 0 | ScaleRowDown34_0(src_ptr, filter_stride, dst_ptr, dst_width); |
521 | 0 | src_ptr += src_stride; |
522 | 0 | dst_ptr += dst_stride; |
523 | 0 | ScaleRowDown34_1(src_ptr, 0, dst_ptr, dst_width); |
524 | 4 | } else if ((dst_height % 3) == 1) { |
525 | 0 | ScaleRowDown34_0(src_ptr, 0, dst_ptr, dst_width); |
526 | 0 | } |
527 | 4 | } |
528 | | |
529 | | // Scale plane, 3/8 |
530 | | // This is an optimized version for scaling down a plane to 3/8 |
531 | | // of its original size. |
532 | | // |
533 | | // Uses box filter arranges like this |
534 | | // aaabbbcc -> abc |
535 | | // aaabbbcc def |
536 | | // aaabbbcc ghi |
537 | | // dddeeeff |
538 | | // dddeeeff |
539 | | // dddeeeff |
540 | | // ggghhhii |
541 | | // ggghhhii |
542 | | // Boxes are 3x3, 2x3, 3x2 and 2x2 |
543 | | |
544 | | static void ScalePlaneDown38(int src_width, |
545 | | int src_height, |
546 | | int dst_width, |
547 | | int dst_height, |
548 | | ptrdiff_t src_stride, |
549 | | ptrdiff_t dst_stride, |
550 | | const uint8_t* src_ptr, |
551 | | uint8_t* dst_ptr, |
552 | 12 | enum FilterMode filtering) { |
553 | 12 | int y; |
554 | 12 | void (*ScaleRowDown38_3)(const uint8_t* src_ptr, ptrdiff_t src_stride, |
555 | 12 | uint8_t* dst_ptr, int dst_width); |
556 | 12 | void (*ScaleRowDown38_2)(const uint8_t* src_ptr, ptrdiff_t src_stride, |
557 | 12 | uint8_t* dst_ptr, int dst_width); |
558 | 12 | const ptrdiff_t filter_stride = (filtering == kFilterLinear) ? 0 : src_stride; |
559 | 12 | assert(dst_width % 3 == 0); |
560 | 12 | (void)src_width; |
561 | 12 | (void)src_height; |
562 | 12 | if (!filtering) { |
563 | 0 | ScaleRowDown38_3 = ScaleRowDown38_C; |
564 | 0 | ScaleRowDown38_2 = ScaleRowDown38_C; |
565 | 12 | } else { |
566 | 12 | ScaleRowDown38_3 = ScaleRowDown38_3_Box_C; |
567 | 12 | ScaleRowDown38_2 = ScaleRowDown38_2_Box_C; |
568 | 12 | } |
569 | | |
570 | | #if defined(HAS_SCALEROWDOWN38_NEON) |
571 | | if (TestCpuFlag(kCpuHasNEON)) { |
572 | | if (!filtering) { |
573 | | ScaleRowDown38_3 = ScaleRowDown38_Any_NEON; |
574 | | ScaleRowDown38_2 = ScaleRowDown38_Any_NEON; |
575 | | } else { |
576 | | ScaleRowDown38_3 = ScaleRowDown38_3_Box_Any_NEON; |
577 | | ScaleRowDown38_2 = ScaleRowDown38_2_Box_Any_NEON; |
578 | | } |
579 | | if (dst_width % 12 == 0) { |
580 | | if (!filtering) { |
581 | | ScaleRowDown38_3 = ScaleRowDown38_NEON; |
582 | | ScaleRowDown38_2 = ScaleRowDown38_NEON; |
583 | | } else { |
584 | | ScaleRowDown38_3 = ScaleRowDown38_3_Box_NEON; |
585 | | ScaleRowDown38_2 = ScaleRowDown38_2_Box_NEON; |
586 | | } |
587 | | } |
588 | | } |
589 | | #endif |
590 | 12 | #if defined(HAS_SCALEROWDOWN38_SSSE3) |
591 | 12 | if (TestCpuFlag(kCpuHasSSSE3)) { |
592 | 12 | if (!filtering) { |
593 | 0 | ScaleRowDown38_3 = ScaleRowDown38_Any_SSSE3; |
594 | 0 | ScaleRowDown38_2 = ScaleRowDown38_Any_SSSE3; |
595 | 12 | } else { |
596 | 12 | ScaleRowDown38_3 = ScaleRowDown38_3_Box_Any_SSSE3; |
597 | 12 | ScaleRowDown38_2 = ScaleRowDown38_2_Box_Any_SSSE3; |
598 | 12 | } |
599 | 12 | if (dst_width % 12 == 0 && !filtering) { |
600 | 0 | ScaleRowDown38_3 = ScaleRowDown38_SSSE3; |
601 | 0 | ScaleRowDown38_2 = ScaleRowDown38_SSSE3; |
602 | 0 | } |
603 | 12 | if (dst_width % 6 == 0 && filtering) { |
604 | 12 | ScaleRowDown38_3 = ScaleRowDown38_3_Box_SSSE3; |
605 | 12 | ScaleRowDown38_2 = ScaleRowDown38_2_Box_SSSE3; |
606 | 12 | } |
607 | 12 | } |
608 | 12 | #endif |
609 | | #if defined(HAS_SCALEROWDOWN38_LSX) |
610 | | if (TestCpuFlag(kCpuHasLSX)) { |
611 | | if (!filtering) { |
612 | | ScaleRowDown38_3 = ScaleRowDown38_Any_LSX; |
613 | | ScaleRowDown38_2 = ScaleRowDown38_Any_LSX; |
614 | | } else { |
615 | | ScaleRowDown38_3 = ScaleRowDown38_3_Box_Any_LSX; |
616 | | ScaleRowDown38_2 = ScaleRowDown38_2_Box_Any_LSX; |
617 | | } |
618 | | if (dst_width % 12 == 0) { |
619 | | if (!filtering) { |
620 | | ScaleRowDown38_3 = ScaleRowDown38_LSX; |
621 | | ScaleRowDown38_2 = ScaleRowDown38_LSX; |
622 | | } else { |
623 | | ScaleRowDown38_3 = ScaleRowDown38_3_Box_LSX; |
624 | | ScaleRowDown38_2 = ScaleRowDown38_2_Box_LSX; |
625 | | } |
626 | | } |
627 | | } |
628 | | #endif |
629 | | #if defined(HAS_SCALEROWDOWN38_RVV) |
630 | | if (TestCpuFlag(kCpuHasRVV)) { |
631 | | if (!filtering) { |
632 | | ScaleRowDown38_3 = ScaleRowDown38_RVV; |
633 | | ScaleRowDown38_2 = ScaleRowDown38_RVV; |
634 | | } else { |
635 | | ScaleRowDown38_3 = ScaleRowDown38_3_Box_RVV; |
636 | | ScaleRowDown38_2 = ScaleRowDown38_2_Box_RVV; |
637 | | } |
638 | | } |
639 | | #endif |
640 | | |
641 | 36 | for (y = 0; y < dst_height - 2; y += 3) { |
642 | 24 | ScaleRowDown38_3(src_ptr, filter_stride, dst_ptr, dst_width); |
643 | 24 | src_ptr += src_stride * 3; |
644 | 24 | dst_ptr += dst_stride; |
645 | 24 | ScaleRowDown38_3(src_ptr, filter_stride, dst_ptr, dst_width); |
646 | 24 | src_ptr += src_stride * 3; |
647 | 24 | dst_ptr += dst_stride; |
648 | 24 | ScaleRowDown38_2(src_ptr, filter_stride, dst_ptr, dst_width); |
649 | 24 | src_ptr += src_stride * 2; |
650 | 24 | dst_ptr += dst_stride; |
651 | 24 | } |
652 | | |
653 | | // Remainder 1 or 2 rows with last row vertically unfiltered |
654 | 12 | if ((dst_height % 3) == 2) { |
655 | 0 | ScaleRowDown38_3(src_ptr, filter_stride, dst_ptr, dst_width); |
656 | 0 | src_ptr += src_stride * 3; |
657 | 0 | dst_ptr += dst_stride; |
658 | 0 | ScaleRowDown38_3(src_ptr, 0, dst_ptr, dst_width); |
659 | 12 | } else if ((dst_height % 3) == 1) { |
660 | 0 | ScaleRowDown38_3(src_ptr, 0, dst_ptr, dst_width); |
661 | 0 | } |
662 | 12 | } |
663 | | |
664 | | static void ScalePlaneDown38_16(int src_width, |
665 | | int src_height, |
666 | | int dst_width, |
667 | | int dst_height, |
668 | | ptrdiff_t src_stride, |
669 | | ptrdiff_t dst_stride, |
670 | | const uint16_t* src_ptr, |
671 | | uint16_t* dst_ptr, |
672 | 12 | enum FilterMode filtering) { |
673 | 12 | int y; |
674 | 12 | void (*ScaleRowDown38_3)(const uint16_t* src_ptr, ptrdiff_t src_stride, |
675 | 12 | uint16_t* dst_ptr, int dst_width); |
676 | 12 | void (*ScaleRowDown38_2)(const uint16_t* src_ptr, ptrdiff_t src_stride, |
677 | 12 | uint16_t* dst_ptr, int dst_width); |
678 | 12 | const ptrdiff_t filter_stride = (filtering == kFilterLinear) ? 0 : src_stride; |
679 | 12 | (void)src_width; |
680 | 12 | (void)src_height; |
681 | 12 | assert(dst_width % 3 == 0); |
682 | 12 | if (!filtering) { |
683 | 0 | ScaleRowDown38_3 = ScaleRowDown38_16_C; |
684 | 0 | ScaleRowDown38_2 = ScaleRowDown38_16_C; |
685 | 12 | } else { |
686 | 12 | ScaleRowDown38_3 = ScaleRowDown38_3_Box_16_C; |
687 | 12 | ScaleRowDown38_2 = ScaleRowDown38_2_Box_16_C; |
688 | 12 | } |
689 | | #if defined(HAS_SCALEROWDOWN38_16_NEON) |
690 | | if (TestCpuFlag(kCpuHasNEON) && (dst_width % 12 == 0)) { |
691 | | if (!filtering) { |
692 | | ScaleRowDown38_3 = ScaleRowDown38_16_NEON; |
693 | | ScaleRowDown38_2 = ScaleRowDown38_16_NEON; |
694 | | } else { |
695 | | ScaleRowDown38_3 = ScaleRowDown38_3_Box_16_NEON; |
696 | | ScaleRowDown38_2 = ScaleRowDown38_2_Box_16_NEON; |
697 | | } |
698 | | } |
699 | | #endif |
700 | | #if defined(HAS_SCALEROWDOWN38_16_SSSE3) |
701 | | if (TestCpuFlag(kCpuHasSSSE3) && (dst_width % 24 == 0)) { |
702 | | if (!filtering) { |
703 | | ScaleRowDown38_3 = ScaleRowDown38_16_SSSE3; |
704 | | ScaleRowDown38_2 = ScaleRowDown38_16_SSSE3; |
705 | | } else { |
706 | | ScaleRowDown38_3 = ScaleRowDown38_3_Box_16_SSSE3; |
707 | | ScaleRowDown38_2 = ScaleRowDown38_2_Box_16_SSSE3; |
708 | | } |
709 | | } |
710 | | #endif |
711 | | |
712 | 36 | for (y = 0; y < dst_height - 2; y += 3) { |
713 | 24 | ScaleRowDown38_3(src_ptr, filter_stride, dst_ptr, dst_width); |
714 | 24 | src_ptr += src_stride * 3; |
715 | 24 | dst_ptr += dst_stride; |
716 | 24 | ScaleRowDown38_3(src_ptr, filter_stride, dst_ptr, dst_width); |
717 | 24 | src_ptr += src_stride * 3; |
718 | 24 | dst_ptr += dst_stride; |
719 | 24 | ScaleRowDown38_2(src_ptr, filter_stride, dst_ptr, dst_width); |
720 | 24 | src_ptr += src_stride * 2; |
721 | 24 | dst_ptr += dst_stride; |
722 | 24 | } |
723 | | |
724 | | // Remainder 1 or 2 rows with last row vertically unfiltered |
725 | 12 | if ((dst_height % 3) == 2) { |
726 | 0 | ScaleRowDown38_3(src_ptr, filter_stride, dst_ptr, dst_width); |
727 | 0 | src_ptr += src_stride * 3; |
728 | 0 | dst_ptr += dst_stride; |
729 | 0 | ScaleRowDown38_3(src_ptr, 0, dst_ptr, dst_width); |
730 | 12 | } else if ((dst_height % 3) == 1) { |
731 | 0 | ScaleRowDown38_3(src_ptr, 0, dst_ptr, dst_width); |
732 | 0 | } |
733 | 12 | } |
734 | | |
735 | 6.72M | #define MIN1(x) ((x) < 1 ? 1 : (x)) |
736 | | |
737 | 4.57M | static __inline uint32_t SumPixels(int iboxwidth, const uint16_t* src_ptr) { |
738 | 4.57M | uint32_t sum = 0u; |
739 | 4.57M | int x; |
740 | 4.57M | assert(iboxwidth > 0); |
741 | 26.7M | for (x = 0; x < iboxwidth; ++x) { |
742 | 22.1M | sum += src_ptr[x]; |
743 | 22.1M | } |
744 | 4.57M | return sum; |
745 | 4.57M | } |
746 | | |
747 | 5.16M | static __inline uint32_t SumPixels_16(int iboxwidth, const uint32_t* src_ptr) { |
748 | 5.16M | uint32_t sum = 0u; |
749 | 5.16M | int x; |
750 | 5.16M | assert(iboxwidth > 0); |
751 | 33.3M | for (x = 0; x < iboxwidth; ++x) { |
752 | 28.1M | sum += src_ptr[x]; |
753 | 28.1M | } |
754 | 5.16M | return sum; |
755 | 5.16M | } |
756 | | |
757 | | static void ScaleAddCols2_C(int dst_width, |
758 | | int boxheight, |
759 | | int x, |
760 | | int dx, |
761 | | const uint16_t* src_ptr, |
762 | 39.6k | uint8_t* dst_ptr) { |
763 | 39.6k | int i; |
764 | 39.6k | int scaletbl[2]; |
765 | 39.6k | int minboxwidth = dx >> 16; |
766 | 39.6k | int boxwidth; |
767 | 39.6k | scaletbl[0] = 65536 / (MIN1(minboxwidth) * boxheight); |
768 | 39.6k | scaletbl[1] = 65536 / (MIN1(minboxwidth + 1) * boxheight); |
769 | 2.84M | for (i = 0; i < dst_width; ++i) { |
770 | 2.80M | int ix = x >> 16; |
771 | 2.80M | x += dx; |
772 | 2.80M | boxwidth = MIN1((x >> 16) - ix); |
773 | 2.80M | int scaletbl_index = boxwidth - minboxwidth; |
774 | 2.80M | assert((scaletbl_index == 0) || (scaletbl_index == 1)); |
775 | 2.80M | *dst_ptr++ = (uint8_t)(SumPixels(boxwidth, src_ptr + ix) * |
776 | 2.80M | scaletbl[scaletbl_index] >> |
777 | 2.80M | 16); |
778 | 2.80M | } |
779 | 39.6k | } |
780 | | |
781 | | static void ScaleAddCols2_16_C(int dst_width, |
782 | | int boxheight, |
783 | | int x, |
784 | | int dx, |
785 | | const uint32_t* src_ptr, |
786 | 49.9k | uint16_t* dst_ptr) { |
787 | 49.9k | int i; |
788 | 49.9k | int scaletbl[2]; |
789 | 49.9k | int minboxwidth = dx >> 16; |
790 | 49.9k | int boxwidth; |
791 | 49.9k | scaletbl[0] = 65536 / (MIN1(minboxwidth) * boxheight); |
792 | 49.9k | scaletbl[1] = 65536 / (MIN1(minboxwidth + 1) * boxheight); |
793 | 3.61M | for (i = 0; i < dst_width; ++i) { |
794 | 3.56M | int ix = x >> 16; |
795 | 3.56M | x += dx; |
796 | 3.56M | boxwidth = MIN1((x >> 16) - ix); |
797 | 3.56M | int scaletbl_index = boxwidth - minboxwidth; |
798 | 3.56M | assert((scaletbl_index == 0) || (scaletbl_index == 1)); |
799 | 3.56M | *dst_ptr++ = |
800 | 3.56M | SumPixels_16(boxwidth, src_ptr + ix) * scaletbl[scaletbl_index] >> 16; |
801 | 3.56M | } |
802 | 49.9k | } |
803 | | |
804 | | static void ScaleAddCols0_C(int dst_width, |
805 | | int boxheight, |
806 | | int x, |
807 | | int dx, |
808 | | const uint16_t* src_ptr, |
809 | 0 | uint8_t* dst_ptr) { |
810 | 0 | int scaleval = 65536 / boxheight; |
811 | 0 | int i; |
812 | 0 | (void)dx; |
813 | 0 | src_ptr += (x >> 16); |
814 | 0 | for (i = 0; i < dst_width; ++i) { |
815 | 0 | *dst_ptr++ = (uint8_t)(src_ptr[i] * scaleval >> 16); |
816 | 0 | } |
817 | 0 | } |
818 | | |
819 | | static void ScaleAddCols1_C(int dst_width, |
820 | | int boxheight, |
821 | | int x, |
822 | | int dx, |
823 | | const uint16_t* src_ptr, |
824 | 17.9k | uint8_t* dst_ptr) { |
825 | 17.9k | int boxwidth = MIN1(dx >> 16); |
826 | 17.9k | int scaleval = 65536 / (boxwidth * boxheight); |
827 | 17.9k | int i; |
828 | 17.9k | x >>= 16; |
829 | 1.79M | for (i = 0; i < dst_width; ++i) { |
830 | 1.77M | *dst_ptr++ = (uint8_t)(SumPixels(boxwidth, src_ptr + x) * scaleval >> 16); |
831 | 1.77M | x += boxwidth; |
832 | 1.77M | } |
833 | 17.9k | } |
834 | | |
835 | | static void ScaleAddCols1_16_C(int dst_width, |
836 | | int boxheight, |
837 | | int x, |
838 | | int dx, |
839 | | const uint32_t* src_ptr, |
840 | 30.2k | uint16_t* dst_ptr) { |
841 | 30.2k | int boxwidth = MIN1(dx >> 16); |
842 | 30.2k | int scaleval = 65536 / (boxwidth * boxheight); |
843 | 30.2k | int i; |
844 | 1.62M | for (i = 0; i < dst_width; ++i) { |
845 | 1.59M | *dst_ptr++ = SumPixels_16(boxwidth, src_ptr + x) * scaleval >> 16; |
846 | 1.59M | x += boxwidth; |
847 | 1.59M | } |
848 | 30.2k | } |
849 | | |
850 | | // Scale plane down to any dimensions, with interpolation. |
851 | | // (boxfilter). |
852 | | // |
853 | | // Same method as SimpleScale, which is fixed point, outputting |
854 | | // one pixel of destination using fixed point (16.16) to step |
855 | | // through source, sampling a box of pixel with simple |
856 | | // averaging. |
857 | | static int ScalePlaneBox(int src_width, |
858 | | int src_height, |
859 | | int dst_width, |
860 | | int dst_height, |
861 | | ptrdiff_t src_stride, |
862 | | ptrdiff_t dst_stride, |
863 | | const uint8_t* src_ptr, |
864 | 1.14k | uint8_t* dst_ptr) { |
865 | 1.14k | int j, k; |
866 | | // Initial source x/y coordinate and step values as 16.16 fixed point. |
867 | 1.14k | int x = 0; |
868 | 1.14k | int y = 0; |
869 | 1.14k | int dx = 0; |
870 | 1.14k | int dy = 0; |
871 | 1.14k | const int max_y = (src_height << 16); |
872 | 1.14k | ScaleSlope(src_width, src_height, dst_width, dst_height, kFilterBox, &x, &y, |
873 | 1.14k | &dx, &dy); |
874 | 1.14k | src_width = Abs(src_width); |
875 | 1.14k | { |
876 | | // Allocate a row buffer of uint16_t. |
877 | 1.14k | align_buffer_64(row16, src_width * 2); |
878 | 1.14k | if (!row16) |
879 | 0 | return 1; |
880 | 1.14k | void (*ScaleAddCols)(int dst_width, int boxheight, int x, int dx, |
881 | 1.14k | const uint16_t* src_ptr, uint8_t* dst_ptr) = |
882 | 1.14k | (dx & 0xffff) ? ScaleAddCols2_C |
883 | 1.14k | : ((dx != 0x10000) ? ScaleAddCols1_C : ScaleAddCols0_C); |
884 | 1.14k | void (*ScaleAddRow)(const uint8_t* src_ptr, uint16_t* dst_ptr, |
885 | 1.14k | int src_width) = ScaleAddRow_C; |
886 | 1.14k | #if defined(HAS_SCALEADDROW_SSE2) |
887 | 1.14k | if (TestCpuFlag(kCpuHasSSE2)) { |
888 | 1.14k | ScaleAddRow = ScaleAddRow_Any_SSE2; |
889 | 1.14k | if (IS_ALIGNED(src_width, 16)) { |
890 | 156 | ScaleAddRow = ScaleAddRow_SSE2; |
891 | 156 | } |
892 | 1.14k | } |
893 | 1.14k | #endif |
894 | 1.14k | #if defined(HAS_SCALEADDROW_AVX2) |
895 | 1.14k | if (TestCpuFlag(kCpuHasAVX2)) { |
896 | 1.14k | ScaleAddRow = ScaleAddRow_Any_AVX2; |
897 | 1.14k | if (IS_ALIGNED(src_width, 32)) { |
898 | 119 | ScaleAddRow = ScaleAddRow_AVX2; |
899 | 119 | } |
900 | 1.14k | } |
901 | 1.14k | #endif |
902 | | #if defined(HAS_SCALEADDROW_NEON) |
903 | | if (TestCpuFlag(kCpuHasNEON)) { |
904 | | ScaleAddRow = ScaleAddRow_Any_NEON; |
905 | | if (IS_ALIGNED(src_width, 16)) { |
906 | | ScaleAddRow = ScaleAddRow_NEON; |
907 | | } |
908 | | } |
909 | | #endif |
910 | | #if defined(HAS_SCALEADDROW_LSX) |
911 | | if (TestCpuFlag(kCpuHasLSX)) { |
912 | | ScaleAddRow = ScaleAddRow_Any_LSX; |
913 | | if (IS_ALIGNED(src_width, 16)) { |
914 | | ScaleAddRow = ScaleAddRow_LSX; |
915 | | } |
916 | | } |
917 | | #endif |
918 | | #if defined(HAS_SCALEADDROW_RVV) |
919 | | if (TestCpuFlag(kCpuHasRVV)) { |
920 | | ScaleAddRow = ScaleAddRow_RVV; |
921 | | } |
922 | | #endif |
923 | | |
924 | 58.7k | for (j = 0; j < dst_height; ++j) { |
925 | 57.6k | int boxheight; |
926 | 57.6k | int iy = y >> 16; |
927 | 57.6k | const uint8_t* src = src_ptr + iy * src_stride; |
928 | 57.6k | y += dy; |
929 | 57.6k | if (y > max_y) { |
930 | 0 | y = max_y; |
931 | 0 | } |
932 | 57.6k | boxheight = MIN1((y >> 16) - iy); |
933 | 57.6k | memset(row16, 0, src_width * 2); |
934 | 705k | for (k = 0; k < boxheight; ++k) { |
935 | 647k | ScaleAddRow(src, (uint16_t*)(row16), src_width); |
936 | 647k | src += src_stride; |
937 | 647k | } |
938 | 57.6k | ScaleAddCols(dst_width, boxheight, x, dx, (uint16_t*)(row16), dst_ptr); |
939 | 57.6k | dst_ptr += dst_stride; |
940 | 57.6k | } |
941 | 1.14k | free_aligned_buffer_64(row16); |
942 | 1.14k | } |
943 | 0 | return 0; |
944 | 1.14k | } |
945 | | |
946 | | static int ScalePlaneBox_16(int src_width, |
947 | | int src_height, |
948 | | int dst_width, |
949 | | int dst_height, |
950 | | ptrdiff_t src_stride, |
951 | | ptrdiff_t dst_stride, |
952 | | const uint16_t* src_ptr, |
953 | 1.75k | uint16_t* dst_ptr) { |
954 | 1.75k | int j, k; |
955 | | // Initial source x/y coordinate and step values as 16.16 fixed point. |
956 | 1.75k | int x = 0; |
957 | 1.75k | int y = 0; |
958 | 1.75k | int dx = 0; |
959 | 1.75k | int dy = 0; |
960 | 1.75k | const int max_y = (src_height << 16); |
961 | 1.75k | ScaleSlope(src_width, src_height, dst_width, dst_height, kFilterBox, &x, &y, |
962 | 1.75k | &dx, &dy); |
963 | 1.75k | src_width = Abs(src_width); |
964 | 1.75k | { |
965 | | // Allocate a row buffer of uint32_t. |
966 | 1.75k | align_buffer_64(row32, src_width * 4); |
967 | 1.75k | if (!row32) |
968 | 0 | return 1; |
969 | 1.75k | void (*ScaleAddCols)(int dst_width, int boxheight, int x, int dx, |
970 | 1.75k | const uint32_t* src_ptr, uint16_t* dst_ptr) = |
971 | 1.75k | (dx & 0xffff) ? ScaleAddCols2_16_C : ScaleAddCols1_16_C; |
972 | 1.75k | void (*ScaleAddRow)(const uint16_t* src_ptr, uint32_t* dst_ptr, |
973 | 1.75k | int src_width) = ScaleAddRow_16_C; |
974 | | |
975 | | #if defined(HAS_SCALEADDROW_16_SSE2) |
976 | | if (TestCpuFlag(kCpuHasSSE2) && IS_ALIGNED(src_width, 16)) { |
977 | | ScaleAddRow = ScaleAddRow_16_SSE2; |
978 | | } |
979 | | #endif |
980 | | |
981 | 81.9k | for (j = 0; j < dst_height; ++j) { |
982 | 80.2k | int boxheight; |
983 | 80.2k | int iy = y >> 16; |
984 | 80.2k | const uint16_t* src = src_ptr + iy * src_stride; |
985 | 80.2k | y += dy; |
986 | 80.2k | if (y > max_y) { |
987 | 0 | y = max_y; |
988 | 0 | } |
989 | 80.2k | boxheight = MIN1((y >> 16) - iy); |
990 | 80.2k | memset(row32, 0, src_width * 4); |
991 | 1.31M | for (k = 0; k < boxheight; ++k) { |
992 | 1.23M | ScaleAddRow(src, (uint32_t*)(row32), src_width); |
993 | 1.23M | src += src_stride; |
994 | 1.23M | } |
995 | 80.2k | ScaleAddCols(dst_width, boxheight, x, dx, (uint32_t*)(row32), dst_ptr); |
996 | 80.2k | dst_ptr += dst_stride; |
997 | 80.2k | } |
998 | 1.75k | free_aligned_buffer_64(row32); |
999 | 1.75k | } |
1000 | 0 | return 0; |
1001 | 1.75k | } |
1002 | | |
1003 | | // Scale plane down with bilinear interpolation. |
1004 | | static int ScalePlaneBilinearDown(int src_width, |
1005 | | int src_height, |
1006 | | int dst_width, |
1007 | | int dst_height, |
1008 | | ptrdiff_t src_stride, |
1009 | | ptrdiff_t dst_stride, |
1010 | | const uint8_t* src_ptr, |
1011 | | uint8_t* dst_ptr, |
1012 | 4.48k | enum FilterMode filtering) { |
1013 | | // Initial source x/y coordinate and step values as 16.16 fixed point. |
1014 | 4.48k | int x = 0; |
1015 | 4.48k | int y = 0; |
1016 | 4.48k | int dx = 0; |
1017 | 4.48k | int dy = 0; |
1018 | | // TODO(fbarchard): Consider not allocating row buffer for kFilterLinear. |
1019 | | // Allocate a row buffer. |
1020 | 4.48k | align_buffer_64(row, src_width); |
1021 | 4.48k | if (!row) |
1022 | 0 | return 1; |
1023 | | |
1024 | 4.48k | const int max_y = (src_height - 1) << 16; |
1025 | 4.48k | int j; |
1026 | 4.48k | void (*ScaleFilterCols)(uint8_t* dst_ptr, const uint8_t* src_ptr, |
1027 | 4.48k | int dst_width, int x, int dx) = |
1028 | 4.48k | (src_width >= 32768) ? ScaleFilterCols64_C : ScaleFilterCols_C; |
1029 | 4.48k | void (*InterpolateRow)(uint8_t* dst_ptr, const uint8_t* src_ptr, |
1030 | 4.48k | ptrdiff_t src_stride, int dst_width, |
1031 | 4.48k | int source_y_fraction) = InterpolateRow_C; |
1032 | 4.48k | ScaleSlope(src_width, src_height, dst_width, dst_height, filtering, &x, &y, |
1033 | 4.48k | &dx, &dy); |
1034 | 4.48k | src_width = Abs(src_width); |
1035 | | |
1036 | 4.48k | #if defined(HAS_INTERPOLATEROW_AVX2) |
1037 | 4.48k | if (TestCpuFlag(kCpuHasAVX2)) { |
1038 | 4.48k | InterpolateRow = InterpolateRow_Any_AVX2; |
1039 | 4.48k | if (IS_ALIGNED(src_width, 32)) { |
1040 | 390 | InterpolateRow = InterpolateRow_AVX2; |
1041 | 390 | } |
1042 | 4.48k | } |
1043 | 4.48k | #endif |
1044 | | #if defined(HAS_INTERPOLATEROW_NEON) |
1045 | | if (TestCpuFlag(kCpuHasNEON)) { |
1046 | | InterpolateRow = InterpolateRow_Any_NEON; |
1047 | | if (IS_ALIGNED(src_width, 16)) { |
1048 | | InterpolateRow = InterpolateRow_NEON; |
1049 | | } |
1050 | | } |
1051 | | #endif |
1052 | | #if defined(HAS_INTERPOLATEROW_SVE2) |
1053 | | if (TestCpuFlag(kCpuHasSVE2)) { |
1054 | | InterpolateRow = InterpolateRow_SVE2; |
1055 | | } |
1056 | | #endif |
1057 | | #if defined(HAS_INTERPOLATEROW_SME) |
1058 | | if (TestCpuFlag(kCpuHasSME)) { |
1059 | | InterpolateRow = InterpolateRow_SME; |
1060 | | } |
1061 | | #endif |
1062 | | #if defined(HAS_INTERPOLATEROW_LSX) |
1063 | | if (TestCpuFlag(kCpuHasLSX)) { |
1064 | | InterpolateRow = InterpolateRow_Any_LSX; |
1065 | | if (IS_ALIGNED(src_width, 32)) { |
1066 | | InterpolateRow = InterpolateRow_LSX; |
1067 | | } |
1068 | | } |
1069 | | #endif |
1070 | | #if defined(HAS_INTERPOLATEROW_RVV) |
1071 | | if (TestCpuFlag(kCpuHasRVV)) { |
1072 | | InterpolateRow = InterpolateRow_RVV; |
1073 | | } |
1074 | | #endif |
1075 | | |
1076 | 4.48k | #if defined(HAS_SCALEFILTERCOLS_SSSE3) |
1077 | 4.48k | if (TestCpuFlag(kCpuHasSSSE3) && src_width < 32768) { |
1078 | 4.48k | ScaleFilterCols = ScaleFilterCols_SSSE3; |
1079 | 4.48k | } |
1080 | 4.48k | #endif |
1081 | | #if defined(HAS_SCALEFILTERCOLS_NEON) |
1082 | | if (TestCpuFlag(kCpuHasNEON) && src_width < 32768) { |
1083 | | ScaleFilterCols = ScaleFilterCols_Any_NEON; |
1084 | | if (IS_ALIGNED(dst_width, 8)) { |
1085 | | ScaleFilterCols = ScaleFilterCols_NEON; |
1086 | | } |
1087 | | } |
1088 | | #endif |
1089 | | #if defined(HAS_SCALEFILTERCOLS_LSX) |
1090 | | if (TestCpuFlag(kCpuHasLSX) && src_width < 32768) { |
1091 | | ScaleFilterCols = ScaleFilterCols_Any_LSX; |
1092 | | if (IS_ALIGNED(dst_width, 16)) { |
1093 | | ScaleFilterCols = ScaleFilterCols_LSX; |
1094 | | } |
1095 | | } |
1096 | | #endif |
1097 | 4.48k | if (y > max_y) { |
1098 | 76 | y = max_y; |
1099 | 76 | } |
1100 | | |
1101 | 200k | for (j = 0; j < dst_height; ++j) { |
1102 | 195k | int yi = y >> 16; |
1103 | 195k | const uint8_t* src = src_ptr + yi * src_stride; |
1104 | 195k | if (filtering == kFilterLinear) { |
1105 | 110k | ScaleFilterCols(dst_ptr, src, dst_width, x, dx); |
1106 | 110k | } else { |
1107 | 85.5k | int yf = (y >> 8) & 255; |
1108 | 85.5k | InterpolateRow(row, src, src_stride, src_width, yf); |
1109 | 85.5k | ScaleFilterCols(dst_ptr, row, dst_width, x, dx); |
1110 | 85.5k | } |
1111 | 195k | dst_ptr += dst_stride; |
1112 | 195k | y += dy; |
1113 | 195k | if (y > max_y) { |
1114 | 6.00k | y = max_y; |
1115 | 6.00k | } |
1116 | 195k | } |
1117 | 4.48k | free_aligned_buffer_64(row); |
1118 | 4.48k | return 0; |
1119 | 4.48k | } |
1120 | | |
1121 | | static int ScalePlaneBilinearDown_16(int src_width, |
1122 | | int src_height, |
1123 | | int dst_width, |
1124 | | int dst_height, |
1125 | | ptrdiff_t src_stride, |
1126 | | ptrdiff_t dst_stride, |
1127 | | const uint16_t* src_ptr, |
1128 | | uint16_t* dst_ptr, |
1129 | 6.98k | enum FilterMode filtering) { |
1130 | | // Initial source x/y coordinate and step values as 16.16 fixed point. |
1131 | 6.98k | int x = 0; |
1132 | 6.98k | int y = 0; |
1133 | 6.98k | int dx = 0; |
1134 | 6.98k | int dy = 0; |
1135 | | // TODO(fbarchard): Consider not allocating row buffer for kFilterLinear. |
1136 | | // Allocate a row buffer. |
1137 | 6.98k | align_buffer_64(row, src_width * 2); |
1138 | 6.98k | if (!row) |
1139 | 0 | return 1; |
1140 | | |
1141 | 6.98k | const int max_y = (src_height - 1) << 16; |
1142 | 6.98k | int j; |
1143 | 6.98k | void (*ScaleFilterCols)(uint16_t* dst_ptr, const uint16_t* src_ptr, |
1144 | 6.98k | int dst_width, int x, int dx) = |
1145 | 6.98k | (src_width >= 32768) ? ScaleFilterCols64_16_C : ScaleFilterCols_16_C; |
1146 | 6.98k | void (*InterpolateRow)(uint16_t* dst_ptr, const uint16_t* src_ptr, |
1147 | 6.98k | ptrdiff_t src_stride, int dst_width, |
1148 | 6.98k | int source_y_fraction) = InterpolateRow_16_C; |
1149 | 6.98k | ScaleSlope(src_width, src_height, dst_width, dst_height, filtering, &x, &y, |
1150 | 6.98k | &dx, &dy); |
1151 | 6.98k | src_width = Abs(src_width); |
1152 | | |
1153 | | #if defined(HAS_INTERPOLATEROW_16_SSSE3) |
1154 | | if (TestCpuFlag(kCpuHasSSSE3)) { |
1155 | | InterpolateRow = InterpolateRow_16_Any_SSSE3; |
1156 | | if (IS_ALIGNED(src_width, 16)) { |
1157 | | InterpolateRow = InterpolateRow_16_SSSE3; |
1158 | | } |
1159 | | } |
1160 | | #endif |
1161 | 6.98k | #if defined(HAS_INTERPOLATEROW_16_AVX2) |
1162 | 6.98k | if (TestCpuFlag(kCpuHasAVX2)) { |
1163 | 6.98k | InterpolateRow = InterpolateRow_16_Any_AVX2; |
1164 | 6.98k | if (IS_ALIGNED(src_width, 32)) { |
1165 | 794 | InterpolateRow = InterpolateRow_16_AVX2; |
1166 | 794 | } |
1167 | 6.98k | } |
1168 | 6.98k | #endif |
1169 | | #if defined(HAS_INTERPOLATEROW_16_NEON) |
1170 | | if (TestCpuFlag(kCpuHasNEON)) { |
1171 | | InterpolateRow = InterpolateRow_16_Any_NEON; |
1172 | | if (IS_ALIGNED(src_width, 16)) { |
1173 | | InterpolateRow = InterpolateRow_16_NEON; |
1174 | | } |
1175 | | } |
1176 | | #endif |
1177 | | #if defined(HAS_INTERPOLATEROW_16_SME) |
1178 | | if (TestCpuFlag(kCpuHasSME)) { |
1179 | | InterpolateRow = InterpolateRow_16_SME; |
1180 | | } |
1181 | | #endif |
1182 | | |
1183 | | #if defined(HAS_SCALEFILTERCOLS_16_SSSE3) |
1184 | | if (TestCpuFlag(kCpuHasSSSE3) && src_width < 32768) { |
1185 | | ScaleFilterCols = ScaleFilterCols_16_SSSE3; |
1186 | | } |
1187 | | #endif |
1188 | 6.98k | if (y > max_y) { |
1189 | 41 | y = max_y; |
1190 | 41 | } |
1191 | | |
1192 | 469k | for (j = 0; j < dst_height; ++j) { |
1193 | 462k | int yi = y >> 16; |
1194 | 462k | const uint16_t* src = src_ptr + yi * src_stride; |
1195 | 462k | if (filtering == kFilterLinear) { |
1196 | 169k | ScaleFilterCols(dst_ptr, src, dst_width, x, dx); |
1197 | 293k | } else { |
1198 | 293k | int yf = (y >> 8) & 255; |
1199 | 293k | InterpolateRow((uint16_t*)row, src, src_stride, src_width, yf); |
1200 | 293k | ScaleFilterCols(dst_ptr, (uint16_t*)row, dst_width, x, dx); |
1201 | 293k | } |
1202 | 462k | dst_ptr += dst_stride; |
1203 | 462k | y += dy; |
1204 | 462k | if (y > max_y) { |
1205 | 8.22k | y = max_y; |
1206 | 8.22k | } |
1207 | 462k | } |
1208 | 6.98k | free_aligned_buffer_64(row); |
1209 | 6.98k | return 0; |
1210 | 6.98k | } |
1211 | | |
1212 | | // Scale up down with bilinear interpolation. |
1213 | | static int ScalePlaneBilinearUp(int src_width, |
1214 | | int src_height, |
1215 | | int dst_width, |
1216 | | int dst_height, |
1217 | | ptrdiff_t src_stride, |
1218 | | ptrdiff_t dst_stride, |
1219 | | const uint8_t* src_ptr, |
1220 | | uint8_t* dst_ptr, |
1221 | 5.42k | enum FilterMode filtering) { |
1222 | 5.42k | assert(src_width > 0); |
1223 | 5.42k | assert(src_height > 0); |
1224 | 5.42k | assert(dst_width > 0); |
1225 | 5.42k | assert(dst_height > 0); |
1226 | | |
1227 | 5.42k | int j; |
1228 | | // Initial source x/y coordinate and step values as 16.16 fixed point. |
1229 | 5.42k | int x = 0; |
1230 | 5.42k | int y = 0; |
1231 | 5.42k | int dx = 0; |
1232 | 5.42k | int dy = 0; |
1233 | 5.42k | const int max_y = (src_height - 1) << 16; |
1234 | 5.42k | void (*InterpolateRow)(uint8_t* dst_ptr, const uint8_t* src_ptr, |
1235 | 5.42k | ptrdiff_t src_stride, int dst_width, |
1236 | 5.42k | int source_y_fraction) = InterpolateRow_C; |
1237 | 5.42k | void (*ScaleFilterCols)(uint8_t* dst_ptr, const uint8_t* src_ptr, |
1238 | 5.42k | int dst_width, int x, int dx) = |
1239 | 5.42k | filtering ? ScaleFilterCols_C : ScaleCols_C; |
1240 | 5.42k | ScaleSlope(src_width, src_height, dst_width, dst_height, filtering, &x, &y, |
1241 | 5.42k | &dx, &dy); |
1242 | 5.42k | assert(dy <= 65536); |
1243 | 5.42k | src_width = Abs(src_width); |
1244 | | |
1245 | 5.42k | #if defined(HAS_INTERPOLATEROW_AVX2) |
1246 | 5.42k | if (TestCpuFlag(kCpuHasAVX2)) { |
1247 | 5.42k | InterpolateRow = InterpolateRow_Any_AVX2; |
1248 | 5.42k | if (IS_ALIGNED(dst_width, 32)) { |
1249 | 2.53k | InterpolateRow = InterpolateRow_AVX2; |
1250 | 2.53k | } |
1251 | 5.42k | } |
1252 | 5.42k | #endif |
1253 | | #if defined(HAS_INTERPOLATEROW_NEON) |
1254 | | if (TestCpuFlag(kCpuHasNEON)) { |
1255 | | InterpolateRow = InterpolateRow_Any_NEON; |
1256 | | if (IS_ALIGNED(dst_width, 16)) { |
1257 | | InterpolateRow = InterpolateRow_NEON; |
1258 | | } |
1259 | | } |
1260 | | #endif |
1261 | | #if defined(HAS_INTERPOLATEROW_SVE2) |
1262 | | if (TestCpuFlag(kCpuHasSVE2)) { |
1263 | | InterpolateRow = InterpolateRow_SVE2; |
1264 | | } |
1265 | | #endif |
1266 | | #if defined(HAS_INTERPOLATEROW_SME) |
1267 | | if (TestCpuFlag(kCpuHasSME)) { |
1268 | | InterpolateRow = InterpolateRow_SME; |
1269 | | } |
1270 | | #endif |
1271 | | #if defined(HAS_INTERPOLATEROW_RVV) |
1272 | | if (TestCpuFlag(kCpuHasRVV)) { |
1273 | | InterpolateRow = InterpolateRow_RVV; |
1274 | | } |
1275 | | #endif |
1276 | | |
1277 | 5.42k | if (filtering && src_width >= 32768) { |
1278 | 0 | ScaleFilterCols = ScaleFilterCols64_C; |
1279 | 0 | } |
1280 | 5.42k | #if defined(HAS_SCALEFILTERCOLS_SSSE3) |
1281 | 5.42k | if (filtering && TestCpuFlag(kCpuHasSSSE3) && src_width < 32768) { |
1282 | 5.42k | ScaleFilterCols = ScaleFilterCols_SSSE3; |
1283 | 5.42k | } |
1284 | 5.42k | #endif |
1285 | | #if defined(HAS_SCALEFILTERCOLS_NEON) |
1286 | | if (filtering && TestCpuFlag(kCpuHasNEON) && src_width < 32768) { |
1287 | | ScaleFilterCols = ScaleFilterCols_Any_NEON; |
1288 | | if (IS_ALIGNED(dst_width, 8)) { |
1289 | | ScaleFilterCols = ScaleFilterCols_NEON; |
1290 | | } |
1291 | | } |
1292 | | #endif |
1293 | | #if defined(HAS_SCALEFILTERCOLS_LSX) |
1294 | | if (filtering && TestCpuFlag(kCpuHasLSX) && src_width < 32768) { |
1295 | | ScaleFilterCols = ScaleFilterCols_Any_LSX; |
1296 | | if (IS_ALIGNED(dst_width, 16)) { |
1297 | | ScaleFilterCols = ScaleFilterCols_LSX; |
1298 | | } |
1299 | | } |
1300 | | #endif |
1301 | 5.42k | if (!filtering && src_width * 2 == dst_width && x < 0x8000) { |
1302 | 0 | ScaleFilterCols = ScaleColsUp2_C; |
1303 | | #if defined(HAS_SCALECOLS_SSE2) |
1304 | | if (TestCpuFlag(kCpuHasSSE2) && IS_ALIGNED(dst_width, 8)) { |
1305 | | ScaleFilterCols = ScaleColsUp2_SSE2; |
1306 | | } |
1307 | | #endif |
1308 | 0 | } |
1309 | | |
1310 | 5.42k | if (y > max_y) { |
1311 | 1.18k | y = max_y; |
1312 | 1.18k | } |
1313 | 5.42k | { |
1314 | 5.42k | int yi = y >> 16; |
1315 | 5.42k | const uint8_t* src = src_ptr + yi * src_stride; |
1316 | | |
1317 | | // Allocate 2 row buffers. |
1318 | 5.42k | const int row_size = (dst_width + 31) & ~31; |
1319 | 5.42k | align_buffer_64(row, row_size * 2); |
1320 | 5.42k | if (!row) |
1321 | 0 | return 1; |
1322 | | |
1323 | 5.42k | uint8_t* rowptr = row; |
1324 | 5.42k | ptrdiff_t rowstride = row_size; |
1325 | 5.42k | int lasty = yi; |
1326 | | |
1327 | 5.42k | ScaleFilterCols(rowptr, src, dst_width, x, dx); |
1328 | 5.42k | if (src_height > 1) { |
1329 | 4.24k | src += src_stride; |
1330 | 4.24k | } |
1331 | 5.42k | ScaleFilterCols(rowptr + rowstride, src, dst_width, x, dx); |
1332 | 5.42k | if (src_height > 2) { |
1333 | 3.97k | src += src_stride; |
1334 | 3.97k | } |
1335 | | |
1336 | | // 2-row rolling buffer: |
1337 | | // rowptr and (rowptr + rowstride) hold the scaled rows for yi and yi + 1. |
1338 | | // Because dy <= 65536 (dy <= 1.0 in 16.16), yi advances in unit steps. |
1339 | | // When yi != lasty: |
1340 | | // 1. Scale the next source row into the older buffer (rowptr). |
1341 | | // 2. Swap buffer pointers (rowptr += rowstride; rowstride = -rowstride;) |
1342 | | // so rowptr points to yi and (rowptr + rowstride) points to yi + 1. |
1343 | | // 3. Advance src by 1 row if row yi + 2 exists ((y + 65536) < max_y), |
1344 | | // otherwise clamp src at (src_height - 1) to avoid reading out of bounds. |
1345 | 2.15M | for (j = 0; j < dst_height; ++j) { |
1346 | 2.14M | if (y > max_y) { |
1347 | 261k | y = max_y; |
1348 | 261k | } |
1349 | 2.14M | yi = y >> 16; |
1350 | 2.14M | if (yi != lasty) { |
1351 | 123k | ScaleFilterCols(rowptr, src, dst_width, x, dx); |
1352 | 123k | rowptr += rowstride; |
1353 | 123k | rowstride = -rowstride; |
1354 | 123k | lasty = yi; |
1355 | 123k | if ((y + 65536) < max_y) { |
1356 | 119k | src += src_stride; |
1357 | 119k | } |
1358 | 123k | } |
1359 | 2.14M | if (filtering == kFilterLinear) { |
1360 | 263k | InterpolateRow(dst_ptr, rowptr, 0, dst_width, 0); |
1361 | 1.88M | } else { |
1362 | 1.88M | int yf = (y >> 8) & 255; |
1363 | 1.88M | InterpolateRow(dst_ptr, rowptr, rowstride, dst_width, yf); |
1364 | 1.88M | } |
1365 | 2.14M | dst_ptr += dst_stride; |
1366 | 2.14M | y += dy; |
1367 | 2.14M | } |
1368 | 5.42k | free_aligned_buffer_64(row); |
1369 | 5.42k | } |
1370 | 0 | return 0; |
1371 | 5.42k | } |
1372 | | |
1373 | | // Scale plane, horizontally up by 2 times. |
1374 | | // Uses linear filter horizontally, nearest vertically. |
1375 | | // This is an optimized version for scaling up a plane to 2 times of |
1376 | | // its original width, using linear interpolation. |
1377 | | // This is used to scale U and V planes of I422 to I444. |
1378 | | static void ScalePlaneUp2_Linear(int src_width, |
1379 | | int src_height, |
1380 | | int dst_width, |
1381 | | int dst_height, |
1382 | | ptrdiff_t src_stride, |
1383 | | ptrdiff_t dst_stride, |
1384 | | const uint8_t* src_ptr, |
1385 | 290 | uint8_t* dst_ptr) { |
1386 | 290 | void (*ScaleRowUp)(const uint8_t* src_ptr, uint8_t* dst_ptr, int dst_width) = |
1387 | 290 | ScaleRowUp2_Linear_Any_C; |
1388 | 290 | int i; |
1389 | 290 | int y; |
1390 | 290 | int dy; |
1391 | | |
1392 | 290 | (void)src_width; |
1393 | | // This function can only scale up by 2 times horizontally. |
1394 | 290 | assert(src_width == ((dst_width + 1) / 2)); |
1395 | | |
1396 | 290 | #ifdef HAS_SCALEROWUP2_LINEAR_SSE2 |
1397 | 290 | if (TestCpuFlag(kCpuHasSSE2)) { |
1398 | 290 | ScaleRowUp = ScaleRowUp2_Linear_Any_SSE2; |
1399 | 290 | } |
1400 | 290 | #endif |
1401 | | |
1402 | 290 | #ifdef HAS_SCALEROWUP2_LINEAR_SSSE3 |
1403 | 290 | if (TestCpuFlag(kCpuHasSSSE3)) { |
1404 | 290 | ScaleRowUp = ScaleRowUp2_Linear_Any_SSSE3; |
1405 | 290 | } |
1406 | 290 | #endif |
1407 | | |
1408 | 290 | #ifdef HAS_SCALEROWUP2_LINEAR_AVX2 |
1409 | 290 | if (TestCpuFlag(kCpuHasAVX2)) { |
1410 | 290 | ScaleRowUp = ScaleRowUp2_Linear_Any_AVX2; |
1411 | 290 | } |
1412 | 290 | #endif |
1413 | | |
1414 | | #ifdef HAS_SCALEROWUP2_LINEAR_NEON |
1415 | | if (TestCpuFlag(kCpuHasNEON)) { |
1416 | | ScaleRowUp = ScaleRowUp2_Linear_Any_NEON; |
1417 | | } |
1418 | | #endif |
1419 | | #ifdef HAS_SCALEROWUP2_LINEAR_RVV |
1420 | | if (TestCpuFlag(kCpuHasRVV)) { |
1421 | | ScaleRowUp = ScaleRowUp2_Linear_RVV; |
1422 | | } |
1423 | | #endif |
1424 | | |
1425 | 290 | if (dst_height == 1) { |
1426 | 31 | ScaleRowUp(src_ptr + ((src_height - 1) / 2) * src_stride, dst_ptr, |
1427 | 31 | dst_width); |
1428 | 259 | } else { |
1429 | 259 | dy = FixedDiv(src_height - 1, dst_height - 1); |
1430 | 259 | y = (1 << 15) - 1; |
1431 | 316k | for (i = 0; i < dst_height; ++i) { |
1432 | 316k | ScaleRowUp(src_ptr + (y >> 16) * src_stride, dst_ptr, dst_width); |
1433 | 316k | dst_ptr += dst_stride; |
1434 | 316k | y += dy; |
1435 | 316k | } |
1436 | 259 | } |
1437 | 290 | } |
1438 | | |
1439 | | // Scale plane, up by 2 times. |
1440 | | // This is an optimized version for scaling up a plane to 2 times of |
1441 | | // its original size, using bilinear interpolation. |
1442 | | // This is used to scale U and V planes of I420 to I444. |
1443 | | static void ScalePlaneUp2_Bilinear(int src_width, |
1444 | | int src_height, |
1445 | | int dst_width, |
1446 | | int dst_height, |
1447 | | ptrdiff_t src_stride, |
1448 | | ptrdiff_t dst_stride, |
1449 | | const uint8_t* src_ptr, |
1450 | 235 | uint8_t* dst_ptr) { |
1451 | 235 | void (*Scale2RowUp)(const uint8_t* src_ptr, ptrdiff_t src_stride, |
1452 | 235 | uint8_t* dst_ptr, ptrdiff_t dst_stride, int dst_width) = |
1453 | 235 | ScaleRowUp2_Bilinear_Any_C; |
1454 | 235 | int x; |
1455 | | |
1456 | 235 | (void)src_width; |
1457 | | // This function can only scale up by 2 times. |
1458 | 235 | assert(src_width == ((dst_width + 1) / 2)); |
1459 | 235 | assert(src_height == ((dst_height + 1) / 2)); |
1460 | | |
1461 | 235 | #ifdef HAS_SCALEROWUP2_BILINEAR_SSE2 |
1462 | 235 | if (TestCpuFlag(kCpuHasSSE2)) { |
1463 | 235 | Scale2RowUp = ScaleRowUp2_Bilinear_Any_SSE2; |
1464 | 235 | } |
1465 | 235 | #endif |
1466 | | |
1467 | 235 | #ifdef HAS_SCALEROWUP2_BILINEAR_SSSE3 |
1468 | 235 | if (TestCpuFlag(kCpuHasSSSE3)) { |
1469 | 235 | Scale2RowUp = ScaleRowUp2_Bilinear_Any_SSSE3; |
1470 | 235 | } |
1471 | 235 | #endif |
1472 | | |
1473 | 235 | #ifdef HAS_SCALEROWUP2_BILINEAR_AVX2 |
1474 | 235 | if (TestCpuFlag(kCpuHasAVX2)) { |
1475 | 235 | Scale2RowUp = ScaleRowUp2_Bilinear_Any_AVX2; |
1476 | 235 | } |
1477 | 235 | #endif |
1478 | | |
1479 | | #ifdef HAS_SCALEROWUP2_BILINEAR_NEON |
1480 | | if (TestCpuFlag(kCpuHasNEON)) { |
1481 | | Scale2RowUp = ScaleRowUp2_Bilinear_Any_NEON; |
1482 | | } |
1483 | | #endif |
1484 | | #ifdef HAS_SCALEROWUP2_BILINEAR_RVV |
1485 | | if (TestCpuFlag(kCpuHasRVV)) { |
1486 | | Scale2RowUp = ScaleRowUp2_Bilinear_RVV; |
1487 | | } |
1488 | | #endif |
1489 | | |
1490 | 235 | Scale2RowUp(src_ptr, 0, dst_ptr, 0, dst_width); |
1491 | 235 | dst_ptr += dst_stride; |
1492 | 8.40k | for (x = 0; x < src_height - 1; ++x) { |
1493 | 8.17k | Scale2RowUp(src_ptr, src_stride, dst_ptr, dst_stride, dst_width); |
1494 | 8.17k | src_ptr += src_stride; |
1495 | | // TODO(fbarchard): Test performance of writing one row of destination at a |
1496 | | // time. |
1497 | 8.17k | dst_ptr += 2 * dst_stride; |
1498 | 8.17k | } |
1499 | 235 | if (!(dst_height & 1)) { |
1500 | 172 | Scale2RowUp(src_ptr, 0, dst_ptr, 0, dst_width); |
1501 | 172 | } |
1502 | 235 | } |
1503 | | |
1504 | | // Scale at most 14 bit plane, horizontally up by 2 times. |
1505 | | // This is an optimized version for scaling up a plane to 2 times of |
1506 | | // its original width, using linear interpolation. |
1507 | | // stride is in count of uint16_t. |
1508 | | // This is used to scale U and V planes of I210 to I410 and I212 to I412. |
1509 | | static void ScalePlaneUp2_12_Linear(int src_width, |
1510 | | int src_height, |
1511 | | int dst_width, |
1512 | | int dst_height, |
1513 | | ptrdiff_t src_stride, |
1514 | | ptrdiff_t dst_stride, |
1515 | | const uint16_t* src_ptr, |
1516 | 283 | uint16_t* dst_ptr) { |
1517 | 283 | void (*ScaleRowUp)(const uint16_t* src_ptr, uint16_t* dst_ptr, |
1518 | 283 | int dst_width) = ScaleRowUp2_Linear_16_Any_C; |
1519 | 283 | int i; |
1520 | 283 | int y; |
1521 | 283 | int dy; |
1522 | | |
1523 | 283 | (void)src_width; |
1524 | | // This function can only scale up by 2 times horizontally. |
1525 | 283 | assert(src_width == ((dst_width + 1) / 2)); |
1526 | | |
1527 | 283 | #ifdef HAS_SCALEROWUP2_LINEAR_12_SSSE3 |
1528 | 283 | if (TestCpuFlag(kCpuHasSSSE3)) { |
1529 | 283 | ScaleRowUp = ScaleRowUp2_Linear_12_Any_SSSE3; |
1530 | 283 | } |
1531 | 283 | #endif |
1532 | | |
1533 | 283 | #ifdef HAS_SCALEROWUP2_LINEAR_12_AVX2 |
1534 | 283 | if (TestCpuFlag(kCpuHasAVX2)) { |
1535 | 283 | ScaleRowUp = ScaleRowUp2_Linear_12_Any_AVX2; |
1536 | 283 | } |
1537 | 283 | #endif |
1538 | | |
1539 | | #ifdef HAS_SCALEROWUP2_LINEAR_12_NEON |
1540 | | if (TestCpuFlag(kCpuHasNEON)) { |
1541 | | ScaleRowUp = ScaleRowUp2_Linear_12_Any_NEON; |
1542 | | } |
1543 | | #endif |
1544 | | |
1545 | 283 | if (dst_height == 1) { |
1546 | 18 | ScaleRowUp(src_ptr + ((src_height - 1) / 2) * src_stride, dst_ptr, |
1547 | 18 | dst_width); |
1548 | 265 | } else { |
1549 | 265 | dy = FixedDiv(src_height - 1, dst_height - 1); |
1550 | 265 | y = (1 << 15) - 1; |
1551 | 620k | for (i = 0; i < dst_height; ++i) { |
1552 | 620k | ScaleRowUp(src_ptr + (y >> 16) * src_stride, dst_ptr, dst_width); |
1553 | 620k | dst_ptr += dst_stride; |
1554 | 620k | y += dy; |
1555 | 620k | } |
1556 | 265 | } |
1557 | 283 | } |
1558 | | |
1559 | | // Scale at most 12 bit plane, up by 2 times. |
1560 | | // This is an optimized version for scaling up a plane to 2 times of |
1561 | | // its original size, using bilinear interpolation. |
1562 | | // stride is in count of uint16_t. |
1563 | | // This is used to scale U and V planes of I010 to I410 and I012 to I412. |
1564 | | static void ScalePlaneUp2_12_Bilinear(int src_width, |
1565 | | int src_height, |
1566 | | int dst_width, |
1567 | | int dst_height, |
1568 | | ptrdiff_t src_stride, |
1569 | | ptrdiff_t dst_stride, |
1570 | | const uint16_t* src_ptr, |
1571 | 178 | uint16_t* dst_ptr) { |
1572 | 178 | void (*Scale2RowUp)(const uint16_t* src_ptr, ptrdiff_t src_stride, |
1573 | 178 | uint16_t* dst_ptr, ptrdiff_t dst_stride, int dst_width) = |
1574 | 178 | ScaleRowUp2_Bilinear_16_Any_C; |
1575 | 178 | int x; |
1576 | | |
1577 | 178 | (void)src_width; |
1578 | | // This function can only scale up by 2 times. |
1579 | 178 | assert(src_width == ((dst_width + 1) / 2)); |
1580 | 178 | assert(src_height == ((dst_height + 1) / 2)); |
1581 | | |
1582 | 178 | #ifdef HAS_SCALEROWUP2_BILINEAR_12_SSSE3 |
1583 | 178 | if (TestCpuFlag(kCpuHasSSSE3)) { |
1584 | 178 | Scale2RowUp = ScaleRowUp2_Bilinear_12_Any_SSSE3; |
1585 | 178 | } |
1586 | 178 | #endif |
1587 | | |
1588 | 178 | #ifdef HAS_SCALEROWUP2_BILINEAR_12_AVX2 |
1589 | 178 | if (TestCpuFlag(kCpuHasAVX2)) { |
1590 | 178 | Scale2RowUp = ScaleRowUp2_Bilinear_12_Any_AVX2; |
1591 | 178 | } |
1592 | 178 | #endif |
1593 | | |
1594 | | #ifdef HAS_SCALEROWUP2_BILINEAR_12_NEON |
1595 | | if (TestCpuFlag(kCpuHasNEON)) { |
1596 | | Scale2RowUp = ScaleRowUp2_Bilinear_12_Any_NEON; |
1597 | | } |
1598 | | #endif |
1599 | | |
1600 | 178 | Scale2RowUp(src_ptr, 0, dst_ptr, 0, dst_width); |
1601 | 178 | dst_ptr += dst_stride; |
1602 | 6.77k | for (x = 0; x < src_height - 1; ++x) { |
1603 | 6.59k | Scale2RowUp(src_ptr, src_stride, dst_ptr, dst_stride, dst_width); |
1604 | 6.59k | src_ptr += src_stride; |
1605 | 6.59k | dst_ptr += 2 * dst_stride; |
1606 | 6.59k | } |
1607 | 178 | if (!(dst_height & 1)) { |
1608 | 113 | Scale2RowUp(src_ptr, 0, dst_ptr, 0, dst_width); |
1609 | 113 | } |
1610 | 178 | } |
1611 | | |
1612 | | static void ScalePlaneUp2_16_Linear(int src_width, |
1613 | | int src_height, |
1614 | | int dst_width, |
1615 | | int dst_height, |
1616 | | ptrdiff_t src_stride, |
1617 | | ptrdiff_t dst_stride, |
1618 | | const uint16_t* src_ptr, |
1619 | 0 | uint16_t* dst_ptr) { |
1620 | 0 | void (*ScaleRowUp)(const uint16_t* src_ptr, uint16_t* dst_ptr, |
1621 | 0 | int dst_width) = ScaleRowUp2_Linear_16_Any_C; |
1622 | 0 | int i; |
1623 | 0 | int y; |
1624 | 0 | int dy; |
1625 | |
|
1626 | 0 | (void)src_width; |
1627 | | // This function can only scale up by 2 times horizontally. |
1628 | 0 | assert(src_width == ((dst_width + 1) / 2)); |
1629 | |
|
1630 | 0 | #ifdef HAS_SCALEROWUP2_LINEAR_16_SSE2 |
1631 | 0 | if (TestCpuFlag(kCpuHasSSE2)) { |
1632 | 0 | ScaleRowUp = ScaleRowUp2_Linear_16_Any_SSE2; |
1633 | 0 | } |
1634 | 0 | #endif |
1635 | |
|
1636 | 0 | #ifdef HAS_SCALEROWUP2_LINEAR_16_AVX2 |
1637 | 0 | if (TestCpuFlag(kCpuHasAVX2)) { |
1638 | 0 | ScaleRowUp = ScaleRowUp2_Linear_16_Any_AVX2; |
1639 | 0 | } |
1640 | 0 | #endif |
1641 | |
|
1642 | | #ifdef HAS_SCALEROWUP2_LINEAR_16_NEON |
1643 | | if (TestCpuFlag(kCpuHasNEON)) { |
1644 | | ScaleRowUp = ScaleRowUp2_Linear_16_Any_NEON; |
1645 | | } |
1646 | | #endif |
1647 | |
|
1648 | 0 | if (dst_height == 1) { |
1649 | 0 | ScaleRowUp(src_ptr + ((src_height - 1) / 2) * src_stride, dst_ptr, |
1650 | 0 | dst_width); |
1651 | 0 | } else { |
1652 | 0 | dy = FixedDiv(src_height - 1, dst_height - 1); |
1653 | 0 | y = (1 << 15) - 1; |
1654 | 0 | for (i = 0; i < dst_height; ++i) { |
1655 | 0 | ScaleRowUp(src_ptr + (y >> 16) * src_stride, dst_ptr, dst_width); |
1656 | 0 | dst_ptr += dst_stride; |
1657 | 0 | y += dy; |
1658 | 0 | } |
1659 | 0 | } |
1660 | 0 | } |
1661 | | |
1662 | | static void ScalePlaneUp2_16_Bilinear(int src_width, |
1663 | | int src_height, |
1664 | | int dst_width, |
1665 | | int dst_height, |
1666 | | ptrdiff_t src_stride, |
1667 | | ptrdiff_t dst_stride, |
1668 | | const uint16_t* src_ptr, |
1669 | 0 | uint16_t* dst_ptr) { |
1670 | 0 | void (*Scale2RowUp)(const uint16_t* src_ptr, ptrdiff_t src_stride, |
1671 | 0 | uint16_t* dst_ptr, ptrdiff_t dst_stride, int dst_width) = |
1672 | 0 | ScaleRowUp2_Bilinear_16_Any_C; |
1673 | 0 | int x; |
1674 | |
|
1675 | 0 | (void)src_width; |
1676 | | // This function can only scale up by 2 times. |
1677 | 0 | assert(src_width == ((dst_width + 1) / 2)); |
1678 | 0 | assert(src_height == ((dst_height + 1) / 2)); |
1679 | |
|
1680 | 0 | #ifdef HAS_SCALEROWUP2_BILINEAR_16_SSE2 |
1681 | 0 | if (TestCpuFlag(kCpuHasSSE2)) { |
1682 | 0 | Scale2RowUp = ScaleRowUp2_Bilinear_16_Any_SSE2; |
1683 | 0 | } |
1684 | 0 | #endif |
1685 | |
|
1686 | 0 | #ifdef HAS_SCALEROWUP2_BILINEAR_16_AVX2 |
1687 | 0 | if (TestCpuFlag(kCpuHasAVX2)) { |
1688 | 0 | Scale2RowUp = ScaleRowUp2_Bilinear_16_Any_AVX2; |
1689 | 0 | } |
1690 | 0 | #endif |
1691 | |
|
1692 | | #ifdef HAS_SCALEROWUP2_BILINEAR_16_NEON |
1693 | | if (TestCpuFlag(kCpuHasNEON)) { |
1694 | | Scale2RowUp = ScaleRowUp2_Bilinear_16_Any_NEON; |
1695 | | } |
1696 | | #endif |
1697 | |
|
1698 | 0 | Scale2RowUp(src_ptr, 0, dst_ptr, 0, dst_width); |
1699 | 0 | dst_ptr += dst_stride; |
1700 | 0 | for (x = 0; x < src_height - 1; ++x) { |
1701 | 0 | Scale2RowUp(src_ptr, src_stride, dst_ptr, dst_stride, dst_width); |
1702 | 0 | src_ptr += src_stride; |
1703 | 0 | dst_ptr += 2 * dst_stride; |
1704 | 0 | } |
1705 | 0 | if (!(dst_height & 1)) { |
1706 | 0 | Scale2RowUp(src_ptr, 0, dst_ptr, 0, dst_width); |
1707 | 0 | } |
1708 | 0 | } |
1709 | | |
1710 | | static int ScalePlaneBilinearUp_16(int src_width, |
1711 | | int src_height, |
1712 | | int dst_width, |
1713 | | int dst_height, |
1714 | | ptrdiff_t src_stride, |
1715 | | ptrdiff_t dst_stride, |
1716 | | const uint16_t* src_ptr, |
1717 | | uint16_t* dst_ptr, |
1718 | 5.85k | enum FilterMode filtering) { |
1719 | 5.85k | assert(src_width > 0); |
1720 | 5.85k | assert(src_height > 0); |
1721 | 5.85k | assert(dst_width > 0); |
1722 | 5.85k | assert(dst_height > 0); |
1723 | | |
1724 | 5.85k | int j; |
1725 | | // Initial source x/y coordinate and step values as 16.16 fixed point. |
1726 | 5.85k | int x = 0; |
1727 | 5.85k | int y = 0; |
1728 | 5.85k | int dx = 0; |
1729 | 5.85k | int dy = 0; |
1730 | 5.85k | const int max_y = (src_height - 1) << 16; |
1731 | 5.85k | void (*InterpolateRow)(uint16_t* dst_ptr, const uint16_t* src_ptr, |
1732 | 5.85k | ptrdiff_t src_stride, int dst_width, |
1733 | 5.85k | int source_y_fraction) = InterpolateRow_16_C; |
1734 | 5.85k | void (*ScaleFilterCols)(uint16_t* dst_ptr, const uint16_t* src_ptr, |
1735 | 5.85k | int dst_width, int x, int dx) = |
1736 | 5.85k | filtering ? ScaleFilterCols_16_C : ScaleCols_16_C; |
1737 | 5.85k | ScaleSlope(src_width, src_height, dst_width, dst_height, filtering, &x, &y, |
1738 | 5.85k | &dx, &dy); |
1739 | 5.85k | assert(dy <= 65536); |
1740 | 5.85k | src_width = Abs(src_width); |
1741 | | |
1742 | | #if defined(HAS_INTERPOLATEROW_16_SSSE3) |
1743 | | if (TestCpuFlag(kCpuHasSSSE3)) { |
1744 | | InterpolateRow = InterpolateRow_16_Any_SSSE3; |
1745 | | if (IS_ALIGNED(dst_width, 16)) { |
1746 | | InterpolateRow = InterpolateRow_16_SSSE3; |
1747 | | } |
1748 | | } |
1749 | | #endif |
1750 | 5.85k | #if defined(HAS_INTERPOLATEROW_16_AVX2) |
1751 | 5.85k | if (TestCpuFlag(kCpuHasAVX2)) { |
1752 | 5.85k | InterpolateRow = InterpolateRow_16_Any_AVX2; |
1753 | 5.85k | if (IS_ALIGNED(dst_width, 32)) { |
1754 | 2.44k | InterpolateRow = InterpolateRow_16_AVX2; |
1755 | 2.44k | } |
1756 | 5.85k | } |
1757 | 5.85k | #endif |
1758 | | #if defined(HAS_INTERPOLATEROW_16_NEON) |
1759 | | if (TestCpuFlag(kCpuHasNEON)) { |
1760 | | InterpolateRow = InterpolateRow_16_Any_NEON; |
1761 | | if (IS_ALIGNED(dst_width, 16)) { |
1762 | | InterpolateRow = InterpolateRow_16_NEON; |
1763 | | } |
1764 | | } |
1765 | | #endif |
1766 | | #if defined(HAS_INTERPOLATEROW_16_SME) |
1767 | | if (TestCpuFlag(kCpuHasSME)) { |
1768 | | InterpolateRow = InterpolateRow_16_SME; |
1769 | | } |
1770 | | #endif |
1771 | | |
1772 | 5.85k | if (filtering && src_width >= 32768) { |
1773 | 0 | ScaleFilterCols = ScaleFilterCols64_16_C; |
1774 | 0 | } |
1775 | | #if defined(HAS_SCALEFILTERCOLS_16_SSSE3) |
1776 | | if (filtering && TestCpuFlag(kCpuHasSSSE3) && src_width < 32768) { |
1777 | | ScaleFilterCols = ScaleFilterCols_16_SSSE3; |
1778 | | } |
1779 | | #endif |
1780 | 5.85k | if (!filtering && src_width * 2 == dst_width && x < 0x8000) { |
1781 | 0 | ScaleFilterCols = ScaleColsUp2_16_C; |
1782 | | #if defined(HAS_SCALECOLS_16_SSE2) |
1783 | | if (TestCpuFlag(kCpuHasSSE2) && IS_ALIGNED(dst_width, 8)) { |
1784 | | ScaleFilterCols = ScaleColsUp2_16_SSE2; |
1785 | | } |
1786 | | #endif |
1787 | 0 | } |
1788 | | |
1789 | 5.85k | if (y > max_y) { |
1790 | 868 | y = max_y; |
1791 | 868 | } |
1792 | 5.85k | { |
1793 | 5.85k | int yi = y >> 16; |
1794 | 5.85k | const uint16_t* src = src_ptr + yi * src_stride; |
1795 | | |
1796 | | // Allocate 2 row buffers. |
1797 | 5.85k | const int row_size = (dst_width + 31) & ~31; |
1798 | 5.85k | align_buffer_64(row, row_size * 4); |
1799 | 5.85k | ptrdiff_t rowstride = row_size; |
1800 | 5.85k | int lasty = yi; |
1801 | 5.85k | uint16_t* rowptr = (uint16_t*)row; |
1802 | 5.85k | if (!row) |
1803 | 0 | return 1; |
1804 | | |
1805 | 5.85k | ScaleFilterCols(rowptr, src, dst_width, x, dx); |
1806 | 5.85k | if (src_height > 1) { |
1807 | 4.98k | src += src_stride; |
1808 | 4.98k | } |
1809 | 5.85k | ScaleFilterCols(rowptr + rowstride, src, dst_width, x, dx); |
1810 | 5.85k | if (src_height > 2) { |
1811 | 4.05k | src += src_stride; |
1812 | 4.05k | } |
1813 | | |
1814 | | // 2-row rolling buffer: |
1815 | | // rowptr and (rowptr + rowstride) hold the scaled rows for yi and yi + 1. |
1816 | | // Because dy <= 65536 (dy <= 1.0 in 16.16), yi advances in unit steps. |
1817 | | // When yi != lasty: |
1818 | | // 1. Scale the next source row into the older buffer (rowptr). |
1819 | | // 2. Swap buffer pointers (rowptr += rowstride; rowstride = -rowstride;) |
1820 | | // so rowptr points to yi and (rowptr + rowstride) points to yi + 1. |
1821 | | // 3. Advance src by 1 row if row yi + 2 exists ((y + 65536) < max_y), |
1822 | | // otherwise clamp src at (src_height - 1) to avoid reading out of bounds. |
1823 | 3.58M | for (j = 0; j < dst_height; ++j) { |
1824 | 3.57M | if (y > max_y) { |
1825 | 323k | y = max_y; |
1826 | 323k | } |
1827 | 3.57M | yi = y >> 16; |
1828 | 3.57M | if (yi != lasty) { |
1829 | 159k | ScaleFilterCols(rowptr, src, dst_width, x, dx); |
1830 | 159k | rowptr += rowstride; |
1831 | 159k | rowstride = -rowstride; |
1832 | 159k | lasty = yi; |
1833 | 159k | if ((y + 65536) < max_y) { |
1834 | 155k | src += src_stride; |
1835 | 155k | } |
1836 | 159k | } |
1837 | 3.57M | if (filtering == kFilterLinear) { |
1838 | 323k | InterpolateRow(dst_ptr, rowptr, 0, dst_width, 0); |
1839 | 3.25M | } else { |
1840 | 3.25M | int yf = (y >> 8) & 255; |
1841 | 3.25M | InterpolateRow(dst_ptr, rowptr, rowstride, dst_width, yf); |
1842 | 3.25M | } |
1843 | 3.57M | dst_ptr += dst_stride; |
1844 | 3.57M | y += dy; |
1845 | 3.57M | } |
1846 | 5.85k | free_aligned_buffer_64(row); |
1847 | 5.85k | } |
1848 | 0 | return 0; |
1849 | 5.85k | } |
1850 | | |
1851 | | // Scale Plane to/from any dimensions, without interpolation. |
1852 | | // Fixed point math is used for performance: The upper 16 bits |
1853 | | // of x and dx is the integer part of the source position and |
1854 | | // the lower 16 bits are the fixed decimal part. |
1855 | | |
1856 | | static void ScalePlaneSimple(int src_width, |
1857 | | int src_height, |
1858 | | int dst_width, |
1859 | | int dst_height, |
1860 | | ptrdiff_t src_stride, |
1861 | | ptrdiff_t dst_stride, |
1862 | | const uint8_t* src_ptr, |
1863 | 1.29k | uint8_t* dst_ptr) { |
1864 | 1.29k | int i; |
1865 | 1.29k | void (*ScaleCols)(uint8_t* dst_ptr, const uint8_t* src_ptr, int dst_width, |
1866 | 1.29k | int x, int dx) = ScaleCols_C; |
1867 | | // Initial source x/y coordinate and step values as 16.16 fixed point. |
1868 | 1.29k | int x = 0; |
1869 | 1.29k | int y = 0; |
1870 | 1.29k | int dx = 0; |
1871 | 1.29k | int dy = 0; |
1872 | 1.29k | ScaleSlope(src_width, src_height, dst_width, dst_height, kFilterNone, &x, &y, |
1873 | 1.29k | &dx, &dy); |
1874 | 1.29k | src_width = Abs(src_width); |
1875 | | |
1876 | 1.29k | if (src_width * 2 == dst_width && x < 0x8000) { |
1877 | 105 | ScaleCols = ScaleColsUp2_C; |
1878 | | #if defined(HAS_SCALECOLS_SSE2) |
1879 | | if (TestCpuFlag(kCpuHasSSE2) && IS_ALIGNED(dst_width, 8)) { |
1880 | | ScaleCols = ScaleColsUp2_SSE2; |
1881 | | } |
1882 | | #endif |
1883 | 105 | } |
1884 | | |
1885 | 781k | for (i = 0; i < dst_height; ++i) { |
1886 | 780k | ScaleCols(dst_ptr, src_ptr + (y >> 16) * src_stride, dst_width, x, dx); |
1887 | 780k | dst_ptr += dst_stride; |
1888 | 780k | y += dy; |
1889 | 780k | } |
1890 | 1.29k | } |
1891 | | |
1892 | | static void ScalePlaneSimple_16(int src_width, |
1893 | | int src_height, |
1894 | | int dst_width, |
1895 | | int dst_height, |
1896 | | ptrdiff_t src_stride, |
1897 | | ptrdiff_t dst_stride, |
1898 | | const uint16_t* src_ptr, |
1899 | 1.06k | uint16_t* dst_ptr) { |
1900 | 1.06k | int i; |
1901 | 1.06k | void (*ScaleCols)(uint16_t* dst_ptr, const uint16_t* src_ptr, int dst_width, |
1902 | 1.06k | int x, int dx) = ScaleCols_16_C; |
1903 | | // Initial source x/y coordinate and step values as 16.16 fixed point. |
1904 | 1.06k | int x = 0; |
1905 | 1.06k | int y = 0; |
1906 | 1.06k | int dx = 0; |
1907 | 1.06k | int dy = 0; |
1908 | 1.06k | ScaleSlope(src_width, src_height, dst_width, dst_height, kFilterNone, &x, &y, |
1909 | 1.06k | &dx, &dy); |
1910 | 1.06k | src_width = Abs(src_width); |
1911 | | |
1912 | 1.06k | if (src_width * 2 == dst_width && x < 0x8000) { |
1913 | 87 | ScaleCols = ScaleColsUp2_16_C; |
1914 | | #if defined(HAS_SCALECOLS_16_SSE2) |
1915 | | if (TestCpuFlag(kCpuHasSSE2) && IS_ALIGNED(dst_width, 8)) { |
1916 | | ScaleCols = ScaleColsUp2_16_SSE2; |
1917 | | } |
1918 | | #endif |
1919 | 87 | } |
1920 | | |
1921 | 899k | for (i = 0; i < dst_height; ++i) { |
1922 | 898k | ScaleCols(dst_ptr, src_ptr + (y >> 16) * src_stride, dst_width, x, dx); |
1923 | 898k | dst_ptr += dst_stride; |
1924 | 898k | y += dy; |
1925 | 898k | } |
1926 | 1.06k | } |
1927 | | |
1928 | | // Scale a plane. |
1929 | | // This function dispatches to a specialized scaler based on scale factor. |
1930 | | LIBYUV_API |
1931 | | int ScalePlane(const uint8_t* src, |
1932 | | int src_stride, |
1933 | | int src_width, |
1934 | | int src_height, |
1935 | | uint8_t* dst, |
1936 | | int dst_stride, |
1937 | | int dst_width, |
1938 | | int dst_height, |
1939 | 15.4k | enum FilterMode filtering) { |
1940 | | // Reject dimensions larger than 32768 (or smaller than -32768 for height). |
1941 | | // This prevents FixedDiv signed integer overflows that can lead to division |
1942 | | // by zero/overflow crashes (SIGFPE on x86) or incorrect step calculations. |
1943 | 15.4k | if (!src || src_width <= 0 || src_height == 0 || src_width > 32768 || |
1944 | 15.4k | src_height < -32768 || src_height > 32768 || !dst || dst_width <= 0 || |
1945 | 15.4k | dst_height <= 0) { |
1946 | 0 | return -1; |
1947 | 0 | } |
1948 | | // Simplify filtering when possible. |
1949 | 15.4k | filtering = ScaleFilterReduce(src_width, src_height, dst_width, dst_height, |
1950 | 15.4k | filtering); |
1951 | | |
1952 | | // Negative height means invert the image. |
1953 | 15.4k | if (src_height < 0) { |
1954 | 0 | src_height = -src_height; |
1955 | 0 | src = src + (src_height - 1) * (ptrdiff_t)src_stride; |
1956 | 0 | src_stride = -src_stride; |
1957 | 0 | } |
1958 | | // Use specialized scales to improve performance for common resolutions. |
1959 | | // For example, all the 1/2 scalings will use ScalePlaneDown2() |
1960 | 15.4k | if (dst_width == src_width && dst_height == src_height) { |
1961 | | // Straight copy. |
1962 | 258 | CopyPlane(src, src_stride, dst, dst_stride, dst_width, dst_height); |
1963 | 258 | return 0; |
1964 | 258 | } |
1965 | 15.1k | if (dst_width == src_width && filtering != kFilterBox) { |
1966 | 2.13k | int dy = 0; |
1967 | 2.13k | int y = 0; |
1968 | | // When scaling down, use the center 2 rows to filter. |
1969 | | // When scaling up, last row of destination uses the last 2 source rows. |
1970 | 2.13k | if (dst_height <= src_height) { |
1971 | 728 | dy = FixedDiv(src_height, dst_height); |
1972 | 728 | y = CENTERSTART(dy, -32768); // Subtract 0.5 (32768) to center filter. |
1973 | 1.40k | } else if (src_height > 1 && dst_height > 1) { |
1974 | 1.26k | dy = FixedDiv1(src_height, dst_height); |
1975 | 1.26k | } |
1976 | | // Arbitrary scale vertically, but unscaled horizontally. |
1977 | 2.13k | ScalePlaneVertical(src_height, dst_width, dst_height, src_stride, |
1978 | 2.13k | dst_stride, src, dst, 0, y, dy, /*bpp=*/1, filtering); |
1979 | 2.13k | return 0; |
1980 | 2.13k | } |
1981 | 13.0k | if (dst_width <= Abs(src_width) && dst_height <= src_height) { |
1982 | | // Scale down. |
1983 | 2.76k | if (4 * dst_width == 3 * src_width && 4 * dst_height == 3 * src_height) { |
1984 | | // optimized, 3/4 |
1985 | 2 | ScalePlaneDown34(src_width, src_height, dst_width, dst_height, src_stride, |
1986 | 2 | dst_stride, src, dst, filtering); |
1987 | 2 | return 0; |
1988 | 2 | } |
1989 | 2.76k | if (2 * dst_width == src_width && 2 * dst_height == src_height) { |
1990 | | // optimized, 1/2 |
1991 | 105 | ScalePlaneDown2(src_width, src_height, dst_width, dst_height, src_stride, |
1992 | 105 | dst_stride, src, dst, filtering); |
1993 | 105 | return 0; |
1994 | 105 | } |
1995 | | // 3/8 rounded up for odd sized chroma height. |
1996 | 2.65k | if (8 * dst_width == 3 * src_width && 8 * dst_height == 3 * src_height) { |
1997 | | // optimized, 3/8 |
1998 | 12 | ScalePlaneDown38(src_width, src_height, dst_width, dst_height, src_stride, |
1999 | 12 | dst_stride, src, dst, filtering); |
2000 | 12 | return 0; |
2001 | 12 | } |
2002 | 2.64k | if (4 * dst_width == src_width && 4 * dst_height == src_height && |
2003 | 52 | (filtering == kFilterBox || filtering == kFilterNone)) { |
2004 | | // optimized, 1/4 |
2005 | 52 | ScalePlaneDown4(src_width, src_height, dst_width, dst_height, src_stride, |
2006 | 52 | dst_stride, src, dst, filtering); |
2007 | 52 | return 0; |
2008 | 52 | } |
2009 | 2.64k | } |
2010 | 12.8k | if (filtering == kFilterBox && dst_height * 2 < src_height) { |
2011 | 1.14k | return ScalePlaneBox(src_width, src_height, dst_width, dst_height, |
2012 | 1.14k | src_stride, dst_stride, src, dst); |
2013 | 1.14k | } |
2014 | 11.7k | if ((dst_width + 1) / 2 == src_width && filtering == kFilterLinear) { |
2015 | 290 | ScalePlaneUp2_Linear(src_width, src_height, dst_width, dst_height, |
2016 | 290 | src_stride, dst_stride, src, dst); |
2017 | 290 | return 0; |
2018 | 290 | } |
2019 | 11.4k | if ((dst_height + 1) / 2 == src_height && (dst_width + 1) / 2 == src_width && |
2020 | 291 | (filtering == kFilterBilinear || filtering == kFilterBox)) { |
2021 | 235 | ScalePlaneUp2_Bilinear(src_width, src_height, dst_width, dst_height, |
2022 | 235 | src_stride, dst_stride, src, dst); |
2023 | 235 | return 0; |
2024 | 235 | } |
2025 | 11.2k | if (filtering && dst_height > src_height) { |
2026 | 5.42k | return ScalePlaneBilinearUp(src_width, src_height, dst_width, dst_height, |
2027 | 5.42k | src_stride, dst_stride, src, dst, filtering); |
2028 | 5.42k | } |
2029 | 5.77k | if (filtering) { |
2030 | 4.48k | return ScalePlaneBilinearDown(src_width, src_height, dst_width, dst_height, |
2031 | 4.48k | src_stride, dst_stride, src, dst, filtering); |
2032 | 4.48k | } |
2033 | 1.29k | ScalePlaneSimple(src_width, src_height, dst_width, dst_height, src_stride, |
2034 | 1.29k | dst_stride, src, dst); |
2035 | 1.29k | return 0; |
2036 | 5.77k | } |
2037 | | |
2038 | | LIBYUV_API |
2039 | | int ScalePlane_16(const uint16_t* src, |
2040 | | int src_stride, |
2041 | | int src_width, |
2042 | | int src_height, |
2043 | | uint16_t* dst, |
2044 | | int dst_stride, |
2045 | | int dst_width, |
2046 | | int dst_height, |
2047 | 17.7k | enum FilterMode filtering) { |
2048 | | // Reject dimensions larger than 32768 (or smaller than -32768 for height). |
2049 | | // This prevents FixedDiv signed integer overflows that can lead to division |
2050 | | // by zero/overflow crashes (SIGFPE on x86) or incorrect step calculations. |
2051 | 17.7k | if (!src || src_width <= 0 || src_height == 0 || src_width > 32768 || |
2052 | 17.7k | src_height < -32768 || src_height > 32768 || !dst || dst_width <= 0 || |
2053 | 17.7k | dst_height <= 0) { |
2054 | 0 | return -1; |
2055 | 0 | } |
2056 | | // Simplify filtering when possible. |
2057 | 17.7k | filtering = ScaleFilterReduce(src_width, src_height, dst_width, dst_height, |
2058 | 17.7k | filtering); |
2059 | | |
2060 | | // Negative height means invert the image. |
2061 | 17.7k | if (src_height < 0) { |
2062 | 0 | src_height = -src_height; |
2063 | 0 | src = src + (src_height - 1) * (ptrdiff_t)src_stride; |
2064 | 0 | src_stride = -src_stride; |
2065 | 0 | } |
2066 | | // Use specialized scales to improve performance for common resolutions. |
2067 | | // For example, all the 1/2 scalings will use ScalePlaneDown2() |
2068 | 17.7k | if (dst_width == src_width && dst_height == src_height) { |
2069 | | // Straight copy. |
2070 | 264 | CopyPlane_16(src, src_stride, dst, dst_stride, dst_width, dst_height); |
2071 | 264 | return 0; |
2072 | 264 | } |
2073 | 17.5k | if (dst_width == src_width && filtering != kFilterBox) { |
2074 | 1.69k | int dy = 0; |
2075 | 1.69k | int y = 0; |
2076 | | // When scaling down, use the center 2 rows to filter. |
2077 | | // When scaling up, last row of destination uses the last 2 source rows. |
2078 | 1.69k | if (dst_height <= src_height) { |
2079 | 592 | dy = FixedDiv(src_height, dst_height); |
2080 | 592 | y = CENTERSTART(dy, -32768); // Subtract 0.5 (32768) to center filter. |
2081 | | // When scaling up, ensure the last row of destination uses the last |
2082 | | // source. Avoid divide by zero for dst_height but will do no scaling |
2083 | | // later. |
2084 | 1.10k | } else if (src_height > 1 && dst_height > 1) { |
2085 | 987 | dy = FixedDiv1(src_height, dst_height); |
2086 | 987 | } |
2087 | | // Arbitrary scale vertically, but unscaled horizontally. |
2088 | 1.69k | ScalePlaneVertical_16(src_height, dst_width, dst_height, src_stride, |
2089 | 1.69k | dst_stride, src, dst, 0, y, dy, /*bpp=*/1, filtering); |
2090 | 1.69k | return 0; |
2091 | 1.69k | } |
2092 | 15.8k | if (dst_width <= Abs(src_width) && dst_height <= src_height) { |
2093 | | // Scale down. |
2094 | 3.71k | if (4 * dst_width == 3 * src_width && 4 * dst_height == 3 * src_height) { |
2095 | | // optimized, 3/4 |
2096 | 4 | ScalePlaneDown34_16(src_width, src_height, dst_width, dst_height, |
2097 | 4 | src_stride, dst_stride, src, dst, filtering); |
2098 | 4 | return 0; |
2099 | 4 | } |
2100 | 3.71k | if (2 * dst_width == src_width && 2 * dst_height == src_height) { |
2101 | | // optimized, 1/2 |
2102 | 93 | ScalePlaneDown2_16(src_width, src_height, dst_width, dst_height, |
2103 | 93 | src_stride, dst_stride, src, dst, filtering); |
2104 | 93 | return 0; |
2105 | 93 | } |
2106 | | // 3/8 rounded up for odd sized chroma height. |
2107 | 3.62k | if (8 * dst_width == 3 * src_width && 8 * dst_height == 3 * src_height) { |
2108 | | // optimized, 3/8 |
2109 | 12 | ScalePlaneDown38_16(src_width, src_height, dst_width, dst_height, |
2110 | 12 | src_stride, dst_stride, src, dst, filtering); |
2111 | 12 | return 0; |
2112 | 12 | } |
2113 | 3.61k | if (4 * dst_width == src_width && 4 * dst_height == src_height && |
2114 | 50 | (filtering == kFilterBox || filtering == kFilterNone)) { |
2115 | | // optimized, 1/4 |
2116 | 50 | ScalePlaneDown4_16(src_width, src_height, dst_width, dst_height, |
2117 | 50 | src_stride, dst_stride, src, dst, filtering); |
2118 | 50 | return 0; |
2119 | 50 | } |
2120 | 3.61k | } |
2121 | 15.6k | if (filtering == kFilterBox && dst_height * 2 < src_height) { |
2122 | 1.75k | return ScalePlaneBox_16(src_width, src_height, dst_width, dst_height, |
2123 | 1.75k | src_stride, dst_stride, src, dst); |
2124 | 1.75k | } |
2125 | 13.9k | if ((dst_width + 1) / 2 == src_width && filtering == kFilterLinear) { |
2126 | 0 | ScalePlaneUp2_16_Linear(src_width, src_height, dst_width, dst_height, |
2127 | 0 | src_stride, dst_stride, src, dst); |
2128 | 0 | return 0; |
2129 | 0 | } |
2130 | 13.9k | if ((dst_height + 1) / 2 == src_height && (dst_width + 1) / 2 == src_width && |
2131 | 30 | (filtering == kFilterBilinear || filtering == kFilterBox)) { |
2132 | 0 | ScalePlaneUp2_16_Bilinear(src_width, src_height, dst_width, dst_height, |
2133 | 0 | src_stride, dst_stride, src, dst); |
2134 | 0 | return 0; |
2135 | 0 | } |
2136 | 13.9k | if (filtering && dst_height > src_height) { |
2137 | 5.85k | return ScalePlaneBilinearUp_16(src_width, src_height, dst_width, dst_height, |
2138 | 5.85k | src_stride, dst_stride, src, dst, filtering); |
2139 | 5.85k | } |
2140 | 8.04k | if (filtering) { |
2141 | 6.98k | return ScalePlaneBilinearDown_16(src_width, src_height, dst_width, |
2142 | 6.98k | dst_height, src_stride, dst_stride, src, |
2143 | 6.98k | dst, filtering); |
2144 | 6.98k | } |
2145 | 1.06k | ScalePlaneSimple_16(src_width, src_height, dst_width, dst_height, src_stride, |
2146 | 1.06k | dst_stride, src, dst); |
2147 | 1.06k | return 0; |
2148 | 8.04k | } |
2149 | | |
2150 | | LIBYUV_API |
2151 | | int ScalePlane_12(const uint16_t* src, |
2152 | | int src_stride, |
2153 | | int src_width, |
2154 | | int src_height, |
2155 | | uint16_t* dst, |
2156 | | int dst_stride, |
2157 | | int dst_width, |
2158 | | int dst_height, |
2159 | 18.2k | enum FilterMode filtering) { |
2160 | | // Reject dimensions larger than 32768 (or smaller than -32768 for height). |
2161 | | // This prevents FixedDiv signed integer overflows that can lead to division |
2162 | | // by zero/overflow crashes (SIGFPE on x86) or incorrect step calculations. |
2163 | 18.2k | if (!src || src_width <= 0 || src_height == 0 || src_width > 32768 || |
2164 | 18.2k | src_height < -32768 || src_height > 32768 || !dst || dst_width <= 0 || |
2165 | 18.2k | dst_height <= 0) { |
2166 | 0 | return -1; |
2167 | 0 | } |
2168 | | // Simplify filtering when possible. |
2169 | 18.2k | filtering = ScaleFilterReduce(src_width, src_height, dst_width, dst_height, |
2170 | 18.2k | filtering); |
2171 | | |
2172 | | // Negative height means invert the image. |
2173 | 18.2k | if (src_height < 0) { |
2174 | 0 | src_height = -src_height; |
2175 | 0 | src = src + (src_height - 1) * (ptrdiff_t)src_stride; |
2176 | 0 | src_stride = -src_stride; |
2177 | 0 | } |
2178 | | |
2179 | 18.2k | if ((dst_width + 1) / 2 == src_width && filtering == kFilterLinear) { |
2180 | 283 | ScalePlaneUp2_12_Linear(src_width, src_height, dst_width, dst_height, |
2181 | 283 | src_stride, dst_stride, src, dst); |
2182 | 283 | return 0; |
2183 | 283 | } |
2184 | 17.9k | if ((dst_height + 1) / 2 == src_height && (dst_width + 1) / 2 == src_width && |
2185 | 255 | (filtering == kFilterBilinear || filtering == kFilterBox)) { |
2186 | 178 | ScalePlaneUp2_12_Bilinear(src_width, src_height, dst_width, dst_height, |
2187 | 178 | src_stride, dst_stride, src, dst); |
2188 | 178 | return 0; |
2189 | 178 | } |
2190 | | |
2191 | 17.7k | return ScalePlane_16(src, src_stride, src_width, src_height, dst, dst_stride, |
2192 | 17.7k | dst_width, dst_height, filtering); |
2193 | 17.9k | } |
2194 | | |
2195 | | // Scale an I420 image. |
2196 | | // This function in turn calls a scaling function for each plane. |
2197 | | |
2198 | | LIBYUV_API |
2199 | | int I420Scale(const uint8_t* src_y, |
2200 | | int src_stride_y, |
2201 | | const uint8_t* src_u, |
2202 | | int src_stride_u, |
2203 | | const uint8_t* src_v, |
2204 | | int src_stride_v, |
2205 | | int src_width, |
2206 | | int src_height, |
2207 | | uint8_t* dst_y, |
2208 | | int dst_stride_y, |
2209 | | uint8_t* dst_u, |
2210 | | int dst_stride_u, |
2211 | | uint8_t* dst_v, |
2212 | | int dst_stride_v, |
2213 | | int dst_width, |
2214 | | int dst_height, |
2215 | 0 | enum FilterMode filtering) { |
2216 | 0 | int r; |
2217 | |
|
2218 | 0 | if (!src_y || !src_u || !src_v || src_width <= 0 || src_height == 0 || |
2219 | 0 | src_height == INT_MIN || !dst_y || !dst_u || !dst_v || dst_width <= 0 || |
2220 | 0 | dst_height <= 0) { |
2221 | 0 | return -1; |
2222 | 0 | } |
2223 | 0 | int src_halfwidth = SUBSAMPLE(src_width, 1, 1); |
2224 | 0 | int src_halfheight = SUBSAMPLE(src_height, 1, 1); |
2225 | 0 | int dst_halfwidth = SUBSAMPLE(dst_width, 1, 1); |
2226 | 0 | int dst_halfheight = SUBSAMPLE(dst_height, 1, 1); |
2227 | |
|
2228 | 0 | r = ScalePlane(src_y, src_stride_y, src_width, src_height, dst_y, |
2229 | 0 | dst_stride_y, dst_width, dst_height, filtering); |
2230 | 0 | if (r != 0) { |
2231 | 0 | return r; |
2232 | 0 | } |
2233 | 0 | r = ScalePlane(src_u, src_stride_u, src_halfwidth, src_halfheight, dst_u, |
2234 | 0 | dst_stride_u, dst_halfwidth, dst_halfheight, filtering); |
2235 | 0 | if (r != 0) { |
2236 | 0 | return r; |
2237 | 0 | } |
2238 | 0 | r = ScalePlane(src_v, src_stride_v, src_halfwidth, src_halfheight, dst_v, |
2239 | 0 | dst_stride_v, dst_halfwidth, dst_halfheight, filtering); |
2240 | 0 | return r; |
2241 | 0 | } |
2242 | | |
2243 | | LIBYUV_API |
2244 | | int I420Scale_16(const uint16_t* src_y, |
2245 | | int src_stride_y, |
2246 | | const uint16_t* src_u, |
2247 | | int src_stride_u, |
2248 | | const uint16_t* src_v, |
2249 | | int src_stride_v, |
2250 | | int src_width, |
2251 | | int src_height, |
2252 | | uint16_t* dst_y, |
2253 | | int dst_stride_y, |
2254 | | uint16_t* dst_u, |
2255 | | int dst_stride_u, |
2256 | | uint16_t* dst_v, |
2257 | | int dst_stride_v, |
2258 | | int dst_width, |
2259 | | int dst_height, |
2260 | 0 | enum FilterMode filtering) { |
2261 | 0 | int r; |
2262 | |
|
2263 | 0 | if (!src_y || !src_u || !src_v || src_width <= 0 || src_height == 0 || |
2264 | 0 | src_height == INT_MIN || !dst_y || !dst_u || !dst_v || dst_width <= 0 || |
2265 | 0 | dst_height <= 0) { |
2266 | 0 | return -1; |
2267 | 0 | } |
2268 | 0 | int src_halfwidth = SUBSAMPLE(src_width, 1, 1); |
2269 | 0 | int src_halfheight = SUBSAMPLE(src_height, 1, 1); |
2270 | 0 | int dst_halfwidth = SUBSAMPLE(dst_width, 1, 1); |
2271 | 0 | int dst_halfheight = SUBSAMPLE(dst_height, 1, 1); |
2272 | |
|
2273 | 0 | r = ScalePlane_16(src_y, src_stride_y, src_width, src_height, dst_y, |
2274 | 0 | dst_stride_y, dst_width, dst_height, filtering); |
2275 | 0 | if (r != 0) { |
2276 | 0 | return r; |
2277 | 0 | } |
2278 | 0 | r = ScalePlane_16(src_u, src_stride_u, src_halfwidth, src_halfheight, dst_u, |
2279 | 0 | dst_stride_u, dst_halfwidth, dst_halfheight, filtering); |
2280 | 0 | if (r != 0) { |
2281 | 0 | return r; |
2282 | 0 | } |
2283 | 0 | r = ScalePlane_16(src_v, src_stride_v, src_halfwidth, src_halfheight, dst_v, |
2284 | 0 | dst_stride_v, dst_halfwidth, dst_halfheight, filtering); |
2285 | 0 | return r; |
2286 | 0 | } |
2287 | | |
2288 | | LIBYUV_API |
2289 | | int I420Scale_12(const uint16_t* src_y, |
2290 | | int src_stride_y, |
2291 | | const uint16_t* src_u, |
2292 | | int src_stride_u, |
2293 | | const uint16_t* src_v, |
2294 | | int src_stride_v, |
2295 | | int src_width, |
2296 | | int src_height, |
2297 | | uint16_t* dst_y, |
2298 | | int dst_stride_y, |
2299 | | uint16_t* dst_u, |
2300 | | int dst_stride_u, |
2301 | | uint16_t* dst_v, |
2302 | | int dst_stride_v, |
2303 | | int dst_width, |
2304 | | int dst_height, |
2305 | 0 | enum FilterMode filtering) { |
2306 | 0 | int r; |
2307 | |
|
2308 | 0 | if (!src_y || !src_u || !src_v || src_width <= 0 || src_height == 0 || |
2309 | 0 | src_height == INT_MIN || !dst_y || !dst_u || !dst_v || dst_width <= 0 || |
2310 | 0 | dst_height <= 0) { |
2311 | 0 | return -1; |
2312 | 0 | } |
2313 | 0 | int src_halfwidth = SUBSAMPLE(src_width, 1, 1); |
2314 | 0 | int src_halfheight = SUBSAMPLE(src_height, 1, 1); |
2315 | 0 | int dst_halfwidth = SUBSAMPLE(dst_width, 1, 1); |
2316 | 0 | int dst_halfheight = SUBSAMPLE(dst_height, 1, 1); |
2317 | |
|
2318 | 0 | r = ScalePlane_12(src_y, src_stride_y, src_width, src_height, dst_y, |
2319 | 0 | dst_stride_y, dst_width, dst_height, filtering); |
2320 | 0 | if (r != 0) { |
2321 | 0 | return r; |
2322 | 0 | } |
2323 | 0 | r = ScalePlane_12(src_u, src_stride_u, src_halfwidth, src_halfheight, dst_u, |
2324 | 0 | dst_stride_u, dst_halfwidth, dst_halfheight, filtering); |
2325 | 0 | if (r != 0) { |
2326 | 0 | return r; |
2327 | 0 | } |
2328 | 0 | r = ScalePlane_12(src_v, src_stride_v, src_halfwidth, src_halfheight, dst_v, |
2329 | 0 | dst_stride_v, dst_halfwidth, dst_halfheight, filtering); |
2330 | 0 | return r; |
2331 | 0 | } |
2332 | | |
2333 | | // Scale an I444 image. |
2334 | | // This function in turn calls a scaling function for each plane. |
2335 | | |
2336 | | LIBYUV_API |
2337 | | int I444Scale(const uint8_t* src_y, |
2338 | | int src_stride_y, |
2339 | | const uint8_t* src_u, |
2340 | | int src_stride_u, |
2341 | | const uint8_t* src_v, |
2342 | | int src_stride_v, |
2343 | | int src_width, |
2344 | | int src_height, |
2345 | | uint8_t* dst_y, |
2346 | | int dst_stride_y, |
2347 | | uint8_t* dst_u, |
2348 | | int dst_stride_u, |
2349 | | uint8_t* dst_v, |
2350 | | int dst_stride_v, |
2351 | | int dst_width, |
2352 | | int dst_height, |
2353 | 0 | enum FilterMode filtering) { |
2354 | 0 | int r; |
2355 | |
|
2356 | 0 | if (!src_y || !src_u || !src_v || src_width <= 0 || src_height == 0 || |
2357 | 0 | src_height == INT_MIN || !dst_y || !dst_u || !dst_v || dst_width <= 0 || |
2358 | 0 | dst_height <= 0) { |
2359 | 0 | return -1; |
2360 | 0 | } |
2361 | | |
2362 | 0 | r = ScalePlane(src_y, src_stride_y, src_width, src_height, dst_y, |
2363 | 0 | dst_stride_y, dst_width, dst_height, filtering); |
2364 | 0 | if (r != 0) { |
2365 | 0 | return r; |
2366 | 0 | } |
2367 | 0 | r = ScalePlane(src_u, src_stride_u, src_width, src_height, dst_u, |
2368 | 0 | dst_stride_u, dst_width, dst_height, filtering); |
2369 | 0 | if (r != 0) { |
2370 | 0 | return r; |
2371 | 0 | } |
2372 | 0 | r = ScalePlane(src_v, src_stride_v, src_width, src_height, dst_v, |
2373 | 0 | dst_stride_v, dst_width, dst_height, filtering); |
2374 | 0 | return r; |
2375 | 0 | } |
2376 | | |
2377 | | LIBYUV_API |
2378 | | int I444Scale_16(const uint16_t* src_y, |
2379 | | int src_stride_y, |
2380 | | const uint16_t* src_u, |
2381 | | int src_stride_u, |
2382 | | const uint16_t* src_v, |
2383 | | int src_stride_v, |
2384 | | int src_width, |
2385 | | int src_height, |
2386 | | uint16_t* dst_y, |
2387 | | int dst_stride_y, |
2388 | | uint16_t* dst_u, |
2389 | | int dst_stride_u, |
2390 | | uint16_t* dst_v, |
2391 | | int dst_stride_v, |
2392 | | int dst_width, |
2393 | | int dst_height, |
2394 | 0 | enum FilterMode filtering) { |
2395 | 0 | int r; |
2396 | |
|
2397 | 0 | if (!src_y || !src_u || !src_v || src_width <= 0 || src_height == 0 || |
2398 | 0 | src_height == INT_MIN || !dst_y || !dst_u || !dst_v || dst_width <= 0 || |
2399 | 0 | dst_height <= 0) { |
2400 | 0 | return -1; |
2401 | 0 | } |
2402 | | |
2403 | 0 | r = ScalePlane_16(src_y, src_stride_y, src_width, src_height, dst_y, |
2404 | 0 | dst_stride_y, dst_width, dst_height, filtering); |
2405 | 0 | if (r != 0) { |
2406 | 0 | return r; |
2407 | 0 | } |
2408 | 0 | r = ScalePlane_16(src_u, src_stride_u, src_width, src_height, dst_u, |
2409 | 0 | dst_stride_u, dst_width, dst_height, filtering); |
2410 | 0 | if (r != 0) { |
2411 | 0 | return r; |
2412 | 0 | } |
2413 | 0 | r = ScalePlane_16(src_v, src_stride_v, src_width, src_height, dst_v, |
2414 | 0 | dst_stride_v, dst_width, dst_height, filtering); |
2415 | 0 | return r; |
2416 | 0 | } |
2417 | | |
2418 | | LIBYUV_API |
2419 | | int I444Scale_12(const uint16_t* src_y, |
2420 | | int src_stride_y, |
2421 | | const uint16_t* src_u, |
2422 | | int src_stride_u, |
2423 | | const uint16_t* src_v, |
2424 | | int src_stride_v, |
2425 | | int src_width, |
2426 | | int src_height, |
2427 | | uint16_t* dst_y, |
2428 | | int dst_stride_y, |
2429 | | uint16_t* dst_u, |
2430 | | int dst_stride_u, |
2431 | | uint16_t* dst_v, |
2432 | | int dst_stride_v, |
2433 | | int dst_width, |
2434 | | int dst_height, |
2435 | 0 | enum FilterMode filtering) { |
2436 | 0 | int r; |
2437 | |
|
2438 | 0 | if (!src_y || !src_u || !src_v || src_width <= 0 || src_height == 0 || |
2439 | 0 | src_height == INT_MIN || !dst_y || !dst_u || !dst_v || dst_width <= 0 || |
2440 | 0 | dst_height <= 0) { |
2441 | 0 | return -1; |
2442 | 0 | } |
2443 | | |
2444 | 0 | r = ScalePlane_12(src_y, src_stride_y, src_width, src_height, dst_y, |
2445 | 0 | dst_stride_y, dst_width, dst_height, filtering); |
2446 | 0 | if (r != 0) { |
2447 | 0 | return r; |
2448 | 0 | } |
2449 | 0 | r = ScalePlane_12(src_u, src_stride_u, src_width, src_height, dst_u, |
2450 | 0 | dst_stride_u, dst_width, dst_height, filtering); |
2451 | 0 | if (r != 0) { |
2452 | 0 | return r; |
2453 | 0 | } |
2454 | 0 | r = ScalePlane_12(src_v, src_stride_v, src_width, src_height, dst_v, |
2455 | 0 | dst_stride_v, dst_width, dst_height, filtering); |
2456 | 0 | return r; |
2457 | 0 | } |
2458 | | |
2459 | | // Scale an I422 image. |
2460 | | // This function in turn calls a scaling function for each plane. |
2461 | | |
2462 | | LIBYUV_API |
2463 | | int I422Scale(const uint8_t* src_y, |
2464 | | int src_stride_y, |
2465 | | const uint8_t* src_u, |
2466 | | int src_stride_u, |
2467 | | const uint8_t* src_v, |
2468 | | int src_stride_v, |
2469 | | int src_width, |
2470 | | int src_height, |
2471 | | uint8_t* dst_y, |
2472 | | int dst_stride_y, |
2473 | | uint8_t* dst_u, |
2474 | | int dst_stride_u, |
2475 | | uint8_t* dst_v, |
2476 | | int dst_stride_v, |
2477 | | int dst_width, |
2478 | | int dst_height, |
2479 | 0 | enum FilterMode filtering) { |
2480 | 0 | int r; |
2481 | |
|
2482 | 0 | if (!src_y || !src_u || !src_v || src_width <= 0 || src_height == 0 || |
2483 | 0 | src_height == INT_MIN || !dst_y || !dst_u || !dst_v || dst_width <= 0 || |
2484 | 0 | dst_height <= 0) { |
2485 | 0 | return -1; |
2486 | 0 | } |
2487 | 0 | int src_halfwidth = SUBSAMPLE(src_width, 1, 1); |
2488 | 0 | int dst_halfwidth = SUBSAMPLE(dst_width, 1, 1); |
2489 | |
|
2490 | 0 | r = ScalePlane(src_y, src_stride_y, src_width, src_height, dst_y, |
2491 | 0 | dst_stride_y, dst_width, dst_height, filtering); |
2492 | 0 | if (r != 0) { |
2493 | 0 | return r; |
2494 | 0 | } |
2495 | 0 | r = ScalePlane(src_u, src_stride_u, src_halfwidth, src_height, dst_u, |
2496 | 0 | dst_stride_u, dst_halfwidth, dst_height, filtering); |
2497 | 0 | if (r != 0) { |
2498 | 0 | return r; |
2499 | 0 | } |
2500 | 0 | r = ScalePlane(src_v, src_stride_v, src_halfwidth, src_height, dst_v, |
2501 | 0 | dst_stride_v, dst_halfwidth, dst_height, filtering); |
2502 | 0 | return r; |
2503 | 0 | } |
2504 | | |
2505 | | LIBYUV_API |
2506 | | int I422Scale_16(const uint16_t* src_y, |
2507 | | int src_stride_y, |
2508 | | const uint16_t* src_u, |
2509 | | int src_stride_u, |
2510 | | const uint16_t* src_v, |
2511 | | int src_stride_v, |
2512 | | int src_width, |
2513 | | int src_height, |
2514 | | uint16_t* dst_y, |
2515 | | int dst_stride_y, |
2516 | | uint16_t* dst_u, |
2517 | | int dst_stride_u, |
2518 | | uint16_t* dst_v, |
2519 | | int dst_stride_v, |
2520 | | int dst_width, |
2521 | | int dst_height, |
2522 | 0 | enum FilterMode filtering) { |
2523 | 0 | int r; |
2524 | |
|
2525 | 0 | if (!src_y || !src_u || !src_v || src_width <= 0 || src_height == 0 || |
2526 | 0 | src_height == INT_MIN || !dst_y || !dst_u || !dst_v || dst_width <= 0 || |
2527 | 0 | dst_height <= 0) { |
2528 | 0 | return -1; |
2529 | 0 | } |
2530 | 0 | int src_halfwidth = SUBSAMPLE(src_width, 1, 1); |
2531 | 0 | int dst_halfwidth = SUBSAMPLE(dst_width, 1, 1); |
2532 | |
|
2533 | 0 | r = ScalePlane_16(src_y, src_stride_y, src_width, src_height, dst_y, |
2534 | 0 | dst_stride_y, dst_width, dst_height, filtering); |
2535 | 0 | if (r != 0) { |
2536 | 0 | return r; |
2537 | 0 | } |
2538 | 0 | r = ScalePlane_16(src_u, src_stride_u, src_halfwidth, src_height, dst_u, |
2539 | 0 | dst_stride_u, dst_halfwidth, dst_height, filtering); |
2540 | 0 | if (r != 0) { |
2541 | 0 | return r; |
2542 | 0 | } |
2543 | 0 | r = ScalePlane_16(src_v, src_stride_v, src_halfwidth, src_height, dst_v, |
2544 | 0 | dst_stride_v, dst_halfwidth, dst_height, filtering); |
2545 | 0 | return r; |
2546 | 0 | } |
2547 | | |
2548 | | LIBYUV_API |
2549 | | int I422Scale_12(const uint16_t* src_y, |
2550 | | int src_stride_y, |
2551 | | const uint16_t* src_u, |
2552 | | int src_stride_u, |
2553 | | const uint16_t* src_v, |
2554 | | int src_stride_v, |
2555 | | int src_width, |
2556 | | int src_height, |
2557 | | uint16_t* dst_y, |
2558 | | int dst_stride_y, |
2559 | | uint16_t* dst_u, |
2560 | | int dst_stride_u, |
2561 | | uint16_t* dst_v, |
2562 | | int dst_stride_v, |
2563 | | int dst_width, |
2564 | | int dst_height, |
2565 | 0 | enum FilterMode filtering) { |
2566 | 0 | int r; |
2567 | |
|
2568 | 0 | if (!src_y || !src_u || !src_v || src_width <= 0 || src_height == 0 || |
2569 | 0 | src_height == INT_MIN || !dst_y || !dst_u || !dst_v || dst_width <= 0 || |
2570 | 0 | dst_height <= 0) { |
2571 | 0 | return -1; |
2572 | 0 | } |
2573 | 0 | int src_halfwidth = SUBSAMPLE(src_width, 1, 1); |
2574 | 0 | int dst_halfwidth = SUBSAMPLE(dst_width, 1, 1); |
2575 | |
|
2576 | 0 | r = ScalePlane_12(src_y, src_stride_y, src_width, src_height, dst_y, |
2577 | 0 | dst_stride_y, dst_width, dst_height, filtering); |
2578 | 0 | if (r != 0) { |
2579 | 0 | return r; |
2580 | 0 | } |
2581 | 0 | r = ScalePlane_12(src_u, src_stride_u, src_halfwidth, src_height, dst_u, |
2582 | 0 | dst_stride_u, dst_halfwidth, dst_height, filtering); |
2583 | 0 | if (r != 0) { |
2584 | 0 | return r; |
2585 | 0 | } |
2586 | 0 | r = ScalePlane_12(src_v, src_stride_v, src_halfwidth, src_height, dst_v, |
2587 | 0 | dst_stride_v, dst_halfwidth, dst_height, filtering); |
2588 | 0 | return r; |
2589 | 0 | } |
2590 | | |
2591 | | // Scale an NV12 image. |
2592 | | // This function in turn calls a scaling function for each plane. |
2593 | | |
2594 | | LIBYUV_API |
2595 | | int NV12Scale(const uint8_t* src_y, |
2596 | | int src_stride_y, |
2597 | | const uint8_t* src_uv, |
2598 | | int src_stride_uv, |
2599 | | int src_width, |
2600 | | int src_height, |
2601 | | uint8_t* dst_y, |
2602 | | int dst_stride_y, |
2603 | | uint8_t* dst_uv, |
2604 | | int dst_stride_uv, |
2605 | | int dst_width, |
2606 | | int dst_height, |
2607 | 0 | enum FilterMode filtering) { |
2608 | 0 | int r; |
2609 | |
|
2610 | 0 | if (!src_y || !src_uv || src_width <= 0 || src_height == 0 || |
2611 | 0 | src_height == INT_MIN || !dst_y || !dst_uv || dst_width <= 0 || |
2612 | 0 | dst_height <= 0) { |
2613 | 0 | return -1; |
2614 | 0 | } |
2615 | 0 | int src_halfwidth = SUBSAMPLE(src_width, 1, 1); |
2616 | 0 | int src_halfheight = SUBSAMPLE(src_height, 1, 1); |
2617 | 0 | int dst_halfwidth = SUBSAMPLE(dst_width, 1, 1); |
2618 | 0 | int dst_halfheight = SUBSAMPLE(dst_height, 1, 1); |
2619 | |
|
2620 | 0 | r = ScalePlane(src_y, src_stride_y, src_width, src_height, dst_y, |
2621 | 0 | dst_stride_y, dst_width, dst_height, filtering); |
2622 | 0 | if (r != 0) { |
2623 | 0 | return r; |
2624 | 0 | } |
2625 | 0 | r = UVScale(src_uv, src_stride_uv, src_halfwidth, src_halfheight, dst_uv, |
2626 | 0 | dst_stride_uv, dst_halfwidth, dst_halfheight, filtering); |
2627 | 0 | return r; |
2628 | 0 | } |
2629 | | |
2630 | | LIBYUV_API |
2631 | | int NV24Scale(const uint8_t* src_y, |
2632 | | int src_stride_y, |
2633 | | const uint8_t* src_uv, |
2634 | | int src_stride_uv, |
2635 | | int src_width, |
2636 | | int src_height, |
2637 | | uint8_t* dst_y, |
2638 | | int dst_stride_y, |
2639 | | uint8_t* dst_uv, |
2640 | | int dst_stride_uv, |
2641 | | int dst_width, |
2642 | | int dst_height, |
2643 | 0 | enum FilterMode filtering) { |
2644 | 0 | int r; |
2645 | |
|
2646 | 0 | if (!src_y || !src_uv || src_width <= 0 || src_height == 0 || |
2647 | 0 | src_height == INT_MIN || !dst_y || !dst_uv || dst_width <= 0 || |
2648 | 0 | dst_height <= 0) { |
2649 | 0 | return -1; |
2650 | 0 | } |
2651 | | |
2652 | 0 | r = ScalePlane(src_y, src_stride_y, src_width, src_height, dst_y, |
2653 | 0 | dst_stride_y, dst_width, dst_height, filtering); |
2654 | 0 | if (r != 0) { |
2655 | 0 | return r; |
2656 | 0 | } |
2657 | 0 | r = UVScale(src_uv, src_stride_uv, src_width, src_height, dst_uv, |
2658 | 0 | dst_stride_uv, dst_width, dst_height, filtering); |
2659 | 0 | return r; |
2660 | 0 | } |
2661 | | |
2662 | | // Deprecated api |
2663 | | LIBYUV_API |
2664 | | int Scale(const uint8_t* src_y, |
2665 | | const uint8_t* src_u, |
2666 | | const uint8_t* src_v, |
2667 | | int src_stride_y, |
2668 | | int src_stride_u, |
2669 | | int src_stride_v, |
2670 | | int src_width, |
2671 | | int src_height, |
2672 | | uint8_t* dst_y, |
2673 | | uint8_t* dst_u, |
2674 | | uint8_t* dst_v, |
2675 | | int dst_stride_y, |
2676 | | int dst_stride_u, |
2677 | | int dst_stride_v, |
2678 | | int dst_width, |
2679 | | int dst_height, |
2680 | 0 | LIBYUV_BOOL interpolate) { |
2681 | 0 | return I420Scale(src_y, src_stride_y, src_u, src_stride_u, src_v, |
2682 | 0 | src_stride_v, src_width, src_height, dst_y, dst_stride_y, |
2683 | 0 | dst_u, dst_stride_u, dst_v, dst_stride_v, dst_width, |
2684 | 0 | dst_height, interpolate ? kFilterBox : kFilterNone); |
2685 | 0 | } |
2686 | | |
2687 | | #ifdef __cplusplus |
2688 | | } // extern "C" |
2689 | | } // namespace libyuv |
2690 | | #endif |