/src/libwebp/sharpyuv/sharpyuv.c
Line | Count | Source |
1 | | // Copyright 2022 Google Inc. All Rights Reserved. |
2 | | // |
3 | | // Use of this source code is governed by a BSD-style license |
4 | | // that can be found in the COPYING file in the root of the source |
5 | | // tree. An additional intellectual property rights grant can be found |
6 | | // in the file PATENTS. All contributing project authors may |
7 | | // be found in the AUTHORS file in the root of the source tree. |
8 | | // ----------------------------------------------------------------------------- |
9 | | // |
10 | | // Sharp RGB to YUV conversion. |
11 | | // |
12 | | // Author: Skal (pascal.massimino@gmail.com) |
13 | | |
14 | | #include "./sharpyuv.h" |
15 | | |
16 | | #include <assert.h> |
17 | | #include <limits.h> |
18 | | #include <stddef.h> |
19 | | #include <stdlib.h> |
20 | | #include <string.h> |
21 | | |
22 | | #include "./sharpyuv_cpu.h" |
23 | | #include "./sharpyuv_dsp.h" |
24 | | #include "./sharpyuv_gamma.h" |
25 | | #include "webp/types.h" |
26 | | |
27 | | //------------------------------------------------------------------------------ |
28 | | |
29 | 0 | int SharpYuvGetVersion(void) { return SHARPYUV_VERSION; } |
30 | | |
31 | | //------------------------------------------------------------------------------ |
32 | | // Sharp RGB->YUV conversion |
33 | | |
34 | | static const int kNumIterations = 4; |
35 | | |
36 | | // Max bit depth so that intermediate calculations fit in 16 bits. |
37 | | static const int kMaxBitDepth = 14; |
38 | | |
39 | | // Returns the precision shift to use based on the input rgb_bit_depth. |
40 | 0 | static int GetPrecisionShift(int rgb_bit_depth) { |
41 | | // Try to add 2 bits of precision if it fits in kMaxBitDepth. Otherwise remove |
42 | | // bits if needed. |
43 | 0 | return ((rgb_bit_depth + 2) <= kMaxBitDepth) ? 2 |
44 | 0 | : (kMaxBitDepth - rgb_bit_depth); |
45 | 0 | } |
46 | | |
47 | | typedef int16_t fixed_t; // signed type with extra precision for UV |
48 | | typedef uint16_t fixed_y_t; // unsigned type with extra precision for W |
49 | | |
50 | | //------------------------------------------------------------------------------ |
51 | | |
52 | 0 | static fixed_y_t clip_bit_depth(int y, int bit_depth) { |
53 | 0 | const int max = (1 << bit_depth) - 1; |
54 | 0 | return (!(y & ~max)) ? (fixed_y_t)y : (y < 0) ? 0 : max; |
55 | 0 | } |
56 | | |
57 | | //------------------------------------------------------------------------------ |
58 | | |
59 | | static uint32_t ScaleDown(uint16_t a, uint16_t b, uint16_t c, uint16_t d, |
60 | | int bit_depth, |
61 | 0 | SharpYuvTransferFunctionType transfer_type) { |
62 | 0 | const uint32_t A = SharpYuvGammaToLinear(a, bit_depth, transfer_type); |
63 | 0 | const uint32_t B = SharpYuvGammaToLinear(b, bit_depth, transfer_type); |
64 | 0 | const uint32_t C = SharpYuvGammaToLinear(c, bit_depth, transfer_type); |
65 | 0 | const uint32_t D = SharpYuvGammaToLinear(d, bit_depth, transfer_type); |
66 | 0 | return SharpYuvLinearToGamma((A + B + C + D + 2) >> 2, bit_depth, |
67 | 0 | transfer_type); |
68 | 0 | } |
69 | | |
70 | | static WEBP_INLINE void UpdateW(const fixed_y_t* src, fixed_y_t* dst, int w, |
71 | | int bit_depth, |
72 | 0 | SharpYuvTransferFunctionType transfer_type) { |
73 | 0 | int i = 0; |
74 | 0 | if (transfer_type == kSharpYuvTransferFunctionSrgb) { |
75 | 0 | SharpYuvUpdateWSrgb(src, dst, w, bit_depth); |
76 | 0 | return; |
77 | 0 | } |
78 | 0 | do { |
79 | 0 | const uint32_t R = |
80 | 0 | SharpYuvGammaToLinear(src[0 * w + i], bit_depth, transfer_type); |
81 | 0 | const uint32_t G = |
82 | 0 | SharpYuvGammaToLinear(src[1 * w + i], bit_depth, transfer_type); |
83 | 0 | const uint32_t B = |
84 | 0 | SharpYuvGammaToLinear(src[2 * w + i], bit_depth, transfer_type); |
85 | 0 | const uint32_t Y = SharpYuvRGBToGray(R, G, B); |
86 | 0 | dst[i] = (fixed_y_t)SharpYuvLinearToGamma(Y, bit_depth, transfer_type); |
87 | 0 | } while (++i < w); |
88 | 0 | } |
89 | | |
90 | | static void UpdateChroma(const fixed_y_t* src1, const fixed_y_t* src2, |
91 | | fixed_t* dst, int uv_w, int bit_depth, |
92 | 0 | SharpYuvTransferFunctionType transfer_type) { |
93 | 0 | int i = 0; |
94 | 0 | if (transfer_type == kSharpYuvTransferFunctionSrgb) { |
95 | 0 | SharpYuvUpdateChromaSrgb(src1, src2, dst, uv_w, bit_depth); |
96 | 0 | return; |
97 | 0 | } |
98 | 0 | do { |
99 | 0 | const int r = |
100 | 0 | ScaleDown(src1[0 * uv_w + 0], src1[0 * uv_w + 1], src2[0 * uv_w + 0], |
101 | 0 | src2[0 * uv_w + 1], bit_depth, transfer_type); |
102 | 0 | const int g = |
103 | 0 | ScaleDown(src1[2 * uv_w + 0], src1[2 * uv_w + 1], src2[2 * uv_w + 0], |
104 | 0 | src2[2 * uv_w + 1], bit_depth, transfer_type); |
105 | 0 | const int b = |
106 | 0 | ScaleDown(src1[4 * uv_w + 0], src1[4 * uv_w + 1], src2[4 * uv_w + 0], |
107 | 0 | src2[4 * uv_w + 1], bit_depth, transfer_type); |
108 | 0 | const int W = SharpYuvRGBToGray(r, g, b); |
109 | 0 | dst[0 * uv_w] = (fixed_t)(r - W); |
110 | 0 | dst[1 * uv_w] = (fixed_t)(g - W); |
111 | 0 | dst[2 * uv_w] = (fixed_t)(b - W); |
112 | 0 | dst += 1; |
113 | 0 | src1 += 2; |
114 | 0 | src2 += 2; |
115 | 0 | } while (++i < uv_w); |
116 | 0 | } |
117 | | |
118 | 0 | static void StoreGray(const fixed_y_t* rgb, fixed_y_t* y, int w) { |
119 | 0 | int i = 0; |
120 | 0 | assert(w > 0); |
121 | 0 | do { |
122 | 0 | y[i] = SharpYuvRGBToGray(rgb[0 * w + i], rgb[1 * w + i], rgb[2 * w + i]); |
123 | 0 | } while (++i < w); |
124 | 0 | } |
125 | | |
126 | | //------------------------------------------------------------------------------ |
127 | | |
128 | 0 | static WEBP_INLINE fixed_y_t Filter2(int A, int B, int W0, int bit_depth) { |
129 | 0 | const int v0 = (A * 3 + B + 2) >> 2; |
130 | 0 | return clip_bit_depth(v0 + W0, bit_depth); |
131 | 0 | } |
132 | | |
133 | | //------------------------------------------------------------------------------ |
134 | | |
135 | 0 | static WEBP_INLINE int Shift(int v, int shift) { |
136 | 0 | return (shift >= 0) ? (v << shift) : (v >> -shift); |
137 | 0 | } |
138 | | |
139 | | static void ImportOneRow(const uint8_t* const r_ptr, const uint8_t* const g_ptr, |
140 | | const uint8_t* const b_ptr, int rgb_step, |
141 | | int rgb_bit_depth, int pic_width, |
142 | 0 | fixed_y_t* const dst) { |
143 | | // Convert the rgb_step from a number of bytes to a number of uint8_t or |
144 | | // uint16_t values depending the bit depth. |
145 | 0 | const int step = (rgb_bit_depth > 8) ? rgb_step / 2 : rgb_step; |
146 | 0 | const int w = (pic_width + 1) & ~1; |
147 | 0 | const int shift = GetPrecisionShift(rgb_bit_depth); |
148 | 0 | const int max_val = (1 << rgb_bit_depth) - 1; |
149 | 0 | int i = 0; |
150 | |
|
151 | 0 | if (rgb_bit_depth == 8) { |
152 | 0 | do { |
153 | 0 | const int off = i * step; |
154 | 0 | dst[i + 0 * w] = Shift(r_ptr[off], shift); |
155 | 0 | dst[i + 1 * w] = Shift(g_ptr[off], shift); |
156 | 0 | dst[i + 2 * w] = Shift(b_ptr[off], shift); |
157 | 0 | } while (++i < pic_width); |
158 | 0 | } else if (rgb_bit_depth < 16) { |
159 | 0 | do { |
160 | 0 | const int off = i * step; |
161 | 0 | int r = ((const uint16_t*)r_ptr)[off]; |
162 | 0 | int g = ((const uint16_t*)g_ptr)[off]; |
163 | 0 | int b = ((const uint16_t*)b_ptr)[off]; |
164 | 0 | dst[i + 0 * w] = Shift(r > max_val ? max_val : r, shift); |
165 | 0 | dst[i + 1 * w] = Shift(g > max_val ? max_val : g, shift); |
166 | 0 | dst[i + 2 * w] = Shift(b > max_val ? max_val : b, shift); |
167 | 0 | } while (++i < pic_width); |
168 | 0 | } else { // rgb_bit_depth == 16 |
169 | 0 | do { |
170 | 0 | const int off = i * step; |
171 | 0 | int r = ((const uint16_t*)r_ptr)[off]; |
172 | 0 | int g = ((const uint16_t*)g_ptr)[off]; |
173 | 0 | int b = ((const uint16_t*)b_ptr)[off]; |
174 | 0 | dst[i + 0 * w] = Shift(r, shift); |
175 | 0 | dst[i + 1 * w] = Shift(g, shift); |
176 | 0 | dst[i + 2 * w] = Shift(b, shift); |
177 | 0 | } while (++i < pic_width); |
178 | 0 | } |
179 | |
|
180 | 0 | if (pic_width & 1) { // replicate rightmost pixel |
181 | 0 | dst[pic_width + 0 * w] = dst[pic_width + 0 * w - 1]; |
182 | 0 | dst[pic_width + 1 * w] = dst[pic_width + 1 * w - 1]; |
183 | 0 | dst[pic_width + 2 * w] = dst[pic_width + 2 * w - 1]; |
184 | 0 | } |
185 | 0 | } |
186 | | |
187 | | static void InterpolateTwoRows(const fixed_y_t* const best_y, |
188 | | const fixed_t* prev_uv, const fixed_t* cur_uv, |
189 | | const fixed_t* next_uv, int w, fixed_y_t* out1, |
190 | 0 | fixed_y_t* out2, int bit_depth) { |
191 | 0 | const int uv_w = w >> 1; |
192 | 0 | const int len = (w - 1) >> 1; // length to filter |
193 | 0 | int k = 3; |
194 | 0 | while (k-- > 0) { // process each R/G/B segments in turn |
195 | | // special boundary case for i==0 |
196 | 0 | out1[0] = Filter2(cur_uv[0], prev_uv[0], best_y[0], bit_depth); |
197 | 0 | out2[0] = Filter2(cur_uv[0], next_uv[0], best_y[w], bit_depth); |
198 | |
|
199 | 0 | SharpYuvFilterRow(cur_uv, prev_uv, len, best_y + 0 + 1, out1 + 1, |
200 | 0 | bit_depth); |
201 | 0 | SharpYuvFilterRow(cur_uv, next_uv, len, best_y + w + 1, out2 + 1, |
202 | 0 | bit_depth); |
203 | | |
204 | | // special boundary case for i == w - 1 when w is even |
205 | 0 | if (!(w & 1)) { |
206 | 0 | out1[w - 1] = Filter2(cur_uv[uv_w - 1], prev_uv[uv_w - 1], |
207 | 0 | best_y[w - 1 + 0], bit_depth); |
208 | 0 | out2[w - 1] = Filter2(cur_uv[uv_w - 1], next_uv[uv_w - 1], |
209 | 0 | best_y[w - 1 + w], bit_depth); |
210 | 0 | } |
211 | 0 | out1 += w; |
212 | 0 | out2 += w; |
213 | 0 | prev_uv += uv_w; |
214 | 0 | cur_uv += uv_w; |
215 | 0 | next_uv += uv_w; |
216 | 0 | } |
217 | 0 | } |
218 | | |
219 | | static int ConvertWRGBToYUV(const fixed_y_t* best_y, const fixed_t* best_uv, |
220 | | uint8_t* y_ptr, int y_stride, uint8_t* u_ptr, |
221 | | int u_stride, uint8_t* v_ptr, int v_stride, |
222 | | int rgb_bit_depth, int yuv_bit_depth, int width, |
223 | | int height, |
224 | 0 | const SharpYuvConversionMatrix* yuv_matrix) { |
225 | 0 | int j; |
226 | 0 | const fixed_t* const best_uv_base = best_uv; |
227 | 0 | const int w = (width + 1) & ~1; |
228 | 0 | const int h = (height + 1) & ~1; |
229 | 0 | const int uv_w = w >> 1; |
230 | 0 | const int uv_h = h >> 1; |
231 | 0 | const int sfix = GetPrecisionShift(rgb_bit_depth); |
232 | |
|
233 | 0 | best_uv = best_uv_base; |
234 | 0 | j = 0; |
235 | 0 | do { |
236 | 0 | SharpYuvConvertRowY(best_y, best_uv, width, uv_w, yuv_matrix->rgb_to_y, |
237 | 0 | sfix, yuv_bit_depth, y_ptr); |
238 | 0 | best_y += w; |
239 | 0 | best_uv += (j & 1) * 3 * uv_w; |
240 | 0 | y_ptr += y_stride; |
241 | 0 | } while (++j < height); |
242 | |
|
243 | 0 | best_uv = best_uv_base; |
244 | 0 | j = 0; |
245 | 0 | do { |
246 | | // Note r, g and b values here are off by W, but a constant offset on all |
247 | | // 3 components doesn't change the value of u and v with a YCbCr matrix. |
248 | 0 | SharpYuvConvertRowUV(best_uv, uv_w, yuv_matrix->rgb_to_u, |
249 | 0 | yuv_matrix->rgb_to_v, sfix, yuv_bit_depth, u_ptr, |
250 | 0 | v_ptr); |
251 | 0 | best_uv += 3 * uv_w; |
252 | 0 | u_ptr += u_stride; |
253 | 0 | v_ptr += v_stride; |
254 | 0 | } while (++j < uv_h); |
255 | 0 | return 1; |
256 | 0 | } |
257 | | |
258 | | //------------------------------------------------------------------------------ |
259 | | // Main function |
260 | | |
261 | 0 | static void* SafeMalloc(uint64_t nmemb, size_t size) { |
262 | 0 | const uint64_t total_size = nmemb * (uint64_t)size; |
263 | 0 | if (total_size != (size_t)total_size) return NULL; |
264 | 0 | return malloc((size_t)total_size); |
265 | 0 | } |
266 | | |
267 | | static int DoSharpArgbToYuv(const uint8_t* r_ptr, const uint8_t* g_ptr, |
268 | | const uint8_t* b_ptr, int rgb_step, int rgb_stride, |
269 | | int rgb_bit_depth, uint8_t* y_ptr, int y_stride, |
270 | | uint8_t* u_ptr, int u_stride, uint8_t* v_ptr, |
271 | | int v_stride, int yuv_bit_depth, int width, |
272 | | int height, |
273 | | const SharpYuvConversionMatrix* yuv_matrix, |
274 | 0 | SharpYuvTransferFunctionType transfer_type) { |
275 | | // we expand the right/bottom border if needed |
276 | 0 | const int w = (width + 1) & ~1; |
277 | 0 | const int h = (height + 1) & ~1; |
278 | 0 | const int uv_w = w >> 1; |
279 | 0 | const int uv_h = h >> 1; |
280 | 0 | const int y_bit_depth = rgb_bit_depth + GetPrecisionShift(rgb_bit_depth); |
281 | 0 | uint64_t prev_diff_y_sum = ~0; |
282 | 0 | int j, iter; |
283 | |
|
284 | 0 | const uint64_t tmp_buffer_size = (uint64_t)w * 3 * 2; |
285 | 0 | const uint64_t best_y_base_size = (uint64_t)w * h; |
286 | 0 | const uint64_t target_y_base_size = (uint64_t)w * h; |
287 | 0 | const uint64_t best_rgb_y_size = (uint64_t)w * 2; |
288 | 0 | const uint64_t best_uv_base_size = (uint64_t)uv_w * 3 * uv_h; |
289 | 0 | const uint64_t target_uv_base_size = (uint64_t)uv_w * 3 * uv_h; |
290 | 0 | const uint64_t best_rgb_uv_size = (uint64_t)uv_w * 3; |
291 | 0 | fixed_y_t* const tmp_buffer = (fixed_y_t*)SafeMalloc( |
292 | 0 | (tmp_buffer_size + best_y_base_size + target_y_base_size + |
293 | 0 | best_rgb_y_size) + |
294 | 0 | (best_uv_base_size + target_uv_base_size + best_rgb_uv_size), |
295 | 0 | sizeof(*tmp_buffer)); |
296 | 0 | fixed_y_t *best_y_base, *target_y_base, *best_rgb_y; |
297 | 0 | fixed_t *best_uv_base, *target_uv_base, *best_rgb_uv; |
298 | 0 | fixed_y_t *best_y, *target_y; |
299 | 0 | fixed_t *best_uv, *target_uv; |
300 | 0 | const uint64_t diff_y_threshold = (uint64_t)(3.0 * w * h); |
301 | 0 | int ok; |
302 | 0 | assert(w > 0); |
303 | 0 | assert(h > 0); |
304 | 0 | assert(sizeof(fixed_y_t) == sizeof(fixed_t)); |
305 | |
|
306 | 0 | if (tmp_buffer == NULL) { |
307 | 0 | ok = 0; |
308 | 0 | goto End; |
309 | 0 | } |
310 | 0 | best_y_base = tmp_buffer + tmp_buffer_size; |
311 | 0 | target_y_base = best_y_base + best_y_base_size; |
312 | 0 | best_rgb_y = target_y_base + target_y_base_size; |
313 | 0 | best_uv_base = (fixed_t*)(best_rgb_y + best_rgb_y_size); |
314 | 0 | target_uv_base = best_uv_base + best_uv_base_size; |
315 | 0 | best_rgb_uv = target_uv_base + target_uv_base_size; |
316 | 0 | best_y = best_y_base; |
317 | 0 | target_y = target_y_base; |
318 | 0 | best_uv = best_uv_base; |
319 | 0 | target_uv = target_uv_base; |
320 | | |
321 | | // Import RGB samples to W/RGB representation. |
322 | 0 | for (j = 0; j < height; j += 2) { |
323 | 0 | const int is_last_row = (j == height - 1); |
324 | 0 | fixed_y_t* const src1 = tmp_buffer + 0 * w; |
325 | 0 | fixed_y_t* const src2 = tmp_buffer + 3 * w; |
326 | | |
327 | | // prepare two rows of input |
328 | 0 | ImportOneRow(r_ptr, g_ptr, b_ptr, rgb_step, rgb_bit_depth, width, src1); |
329 | 0 | if (!is_last_row) { |
330 | 0 | ImportOneRow(r_ptr + rgb_stride, g_ptr + rgb_stride, b_ptr + rgb_stride, |
331 | 0 | rgb_step, rgb_bit_depth, width, src2); |
332 | 0 | } else { |
333 | 0 | memcpy(src2, src1, 3 * w * sizeof(*src2)); |
334 | 0 | } |
335 | 0 | StoreGray(src1, best_y + 0, w); |
336 | 0 | StoreGray(src2, best_y + w, w); |
337 | |
|
338 | 0 | UpdateW(src1, target_y, w, y_bit_depth, transfer_type); |
339 | 0 | UpdateW(src2, target_y + w, w, y_bit_depth, transfer_type); |
340 | 0 | UpdateChroma(src1, src2, target_uv, uv_w, y_bit_depth, transfer_type); |
341 | 0 | memcpy(best_uv, target_uv, 3 * uv_w * sizeof(*best_uv)); |
342 | 0 | best_y += 2 * w; |
343 | 0 | best_uv += 3 * uv_w; |
344 | 0 | target_y += 2 * w; |
345 | 0 | target_uv += 3 * uv_w; |
346 | 0 | r_ptr += 2 * rgb_stride; |
347 | 0 | g_ptr += 2 * rgb_stride; |
348 | 0 | b_ptr += 2 * rgb_stride; |
349 | 0 | } |
350 | | |
351 | | // Iterate and resolve clipping conflicts. |
352 | 0 | for (iter = 0; iter < kNumIterations; ++iter) { |
353 | 0 | const fixed_t* cur_uv = best_uv_base; |
354 | 0 | const fixed_t* prev_uv = best_uv_base; |
355 | 0 | uint64_t diff_y_sum = 0; |
356 | |
|
357 | 0 | best_y = best_y_base; |
358 | 0 | best_uv = best_uv_base; |
359 | 0 | target_y = target_y_base; |
360 | 0 | target_uv = target_uv_base; |
361 | 0 | j = 0; |
362 | 0 | do { |
363 | 0 | fixed_y_t* const src1 = tmp_buffer + 0 * w; |
364 | 0 | fixed_y_t* const src2 = tmp_buffer + 3 * w; |
365 | 0 | { |
366 | 0 | const fixed_t* const next_uv = cur_uv + ((j < h - 2) ? 3 * uv_w : 0); |
367 | 0 | InterpolateTwoRows(best_y, prev_uv, cur_uv, next_uv, w, src1, src2, |
368 | 0 | y_bit_depth); |
369 | 0 | prev_uv = cur_uv; |
370 | 0 | cur_uv = next_uv; |
371 | 0 | } |
372 | |
|
373 | 0 | UpdateW(src1, best_rgb_y + 0 * w, w, y_bit_depth, transfer_type); |
374 | 0 | UpdateW(src2, best_rgb_y + 1 * w, w, y_bit_depth, transfer_type); |
375 | 0 | UpdateChroma(src1, src2, best_rgb_uv, uv_w, y_bit_depth, transfer_type); |
376 | | |
377 | | // update two rows of Y and one row of RGB |
378 | 0 | diff_y_sum += |
379 | 0 | SharpYuvUpdateY(target_y, best_rgb_y, best_y, 2 * w, y_bit_depth); |
380 | 0 | SharpYuvUpdateRGB(target_uv, best_rgb_uv, best_uv, 3 * uv_w); |
381 | |
|
382 | 0 | best_y += 2 * w; |
383 | 0 | best_uv += 3 * uv_w; |
384 | 0 | target_y += 2 * w; |
385 | 0 | target_uv += 3 * uv_w; |
386 | 0 | j += 2; |
387 | 0 | } while (j < h); |
388 | | // test exit condition |
389 | 0 | if (diff_y_sum < diff_y_threshold) break; |
390 | 0 | if (iter > 0 && diff_y_sum > prev_diff_y_sum) break; |
391 | 0 | prev_diff_y_sum = diff_y_sum; |
392 | 0 | } |
393 | | |
394 | | // final reconstruction |
395 | 0 | ok = ConvertWRGBToYUV(best_y_base, best_uv_base, y_ptr, y_stride, u_ptr, |
396 | 0 | u_stride, v_ptr, v_stride, rgb_bit_depth, yuv_bit_depth, |
397 | 0 | width, height, yuv_matrix); |
398 | |
|
399 | 0 | End: |
400 | 0 | free(tmp_buffer); |
401 | 0 | return ok; |
402 | 0 | } |
403 | | |
404 | | #if defined(WEBP_USE_THREAD) && !defined(_WIN32) |
405 | | #include <pthread.h> // NOLINT |
406 | | |
407 | | #define LOCK_ACCESS \ |
408 | 0 | static pthread_mutex_t sharpyuv_lock = PTHREAD_MUTEX_INITIALIZER; \ |
409 | 0 | if (pthread_mutex_lock(&sharpyuv_lock)) return |
410 | | #define UNLOCK_ACCESS_AND_RETURN \ |
411 | 0 | do { \ |
412 | 0 | (void)pthread_mutex_unlock(&sharpyuv_lock); \ |
413 | 0 | return; \ |
414 | 0 | } while (0) |
415 | | #else // !(defined(WEBP_USE_THREAD) && !defined(_WIN32)) |
416 | | #define LOCK_ACCESS \ |
417 | | do { \ |
418 | | } while (0) |
419 | | #define UNLOCK_ACCESS_AND_RETURN return |
420 | | #endif // defined(WEBP_USE_THREAD) && !defined(_WIN32) |
421 | | |
422 | | // Hidden exported init function. |
423 | | // By default SharpYuvConvert calls it with SharpYuvGetCPUInfo. If needed, |
424 | | // users can declare it as extern and call it with an alternate VP8CPUInfo |
425 | | // function. |
426 | | extern VP8CPUInfo SharpYuvGetCPUInfo; |
427 | | SHARPYUV_EXTERN void SharpYuvInit(VP8CPUInfo cpu_info_func); |
428 | 0 | void SharpYuvInit(VP8CPUInfo cpu_info_func) { |
429 | 0 | static volatile VP8CPUInfo sharpyuv_last_cpuinfo_used = |
430 | 0 | (VP8CPUInfo)&sharpyuv_last_cpuinfo_used; |
431 | 0 | LOCK_ACCESS; |
432 | | // Only update SharpYuvGetCPUInfo when called from external code to avoid a |
433 | | // race on reading the value in SharpYuvConvert(). |
434 | 0 | if (cpu_info_func != (VP8CPUInfo)&SharpYuvGetCPUInfo) { |
435 | 0 | SharpYuvGetCPUInfo = cpu_info_func; |
436 | 0 | } |
437 | 0 | if (sharpyuv_last_cpuinfo_used == SharpYuvGetCPUInfo) { |
438 | 0 | UNLOCK_ACCESS_AND_RETURN; |
439 | 0 | } |
440 | | |
441 | 0 | SharpYuvInitDsp(); |
442 | 0 | SharpYuvInitGammaTables(); |
443 | |
|
444 | 0 | sharpyuv_last_cpuinfo_used = SharpYuvGetCPUInfo; |
445 | 0 | UNLOCK_ACCESS_AND_RETURN; |
446 | 0 | } |
447 | | |
448 | | int SharpYuvConvert(const void* r_ptr, const void* g_ptr, const void* b_ptr, |
449 | | int rgb_step, int rgb_stride, int rgb_bit_depth, |
450 | | void* y_ptr, int y_stride, void* u_ptr, int u_stride, |
451 | | void* v_ptr, int v_stride, int yuv_bit_depth, int width, |
452 | 0 | int height, const SharpYuvConversionMatrix* yuv_matrix) { |
453 | 0 | SharpYuvOptions options; |
454 | 0 | options.yuv_matrix = yuv_matrix; |
455 | 0 | options.transfer_type = kSharpYuvTransferFunctionSrgb; |
456 | 0 | return SharpYuvConvertWithOptions( |
457 | 0 | r_ptr, g_ptr, b_ptr, rgb_step, rgb_stride, rgb_bit_depth, y_ptr, y_stride, |
458 | 0 | u_ptr, u_stride, v_ptr, v_stride, yuv_bit_depth, width, height, &options); |
459 | 0 | } |
460 | | |
461 | | int SharpYuvOptionsInitInternal(const SharpYuvConversionMatrix* yuv_matrix, |
462 | 0 | SharpYuvOptions* options, int version) { |
463 | 0 | const int major = (version >> 24); |
464 | 0 | const int minor = (version >> 16) & 0xff; |
465 | 0 | if (options == NULL || yuv_matrix == NULL || |
466 | 0 | (major == SHARPYUV_VERSION_MAJOR && major == 0 && |
467 | 0 | minor != SHARPYUV_VERSION_MINOR) || |
468 | 0 | (major != SHARPYUV_VERSION_MAJOR)) { |
469 | 0 | return 0; |
470 | 0 | } |
471 | 0 | options->yuv_matrix = yuv_matrix; |
472 | 0 | options->transfer_type = kSharpYuvTransferFunctionSrgb; |
473 | 0 | return 1; |
474 | 0 | } |
475 | | |
476 | | int SharpYuvConvertWithOptions(const void* r_ptr, const void* g_ptr, |
477 | | const void* b_ptr, int rgb_step, int rgb_stride, |
478 | | int rgb_bit_depth, void* y_ptr, int y_stride, |
479 | | void* u_ptr, int u_stride, void* v_ptr, |
480 | | int v_stride, int yuv_bit_depth, int width, |
481 | 0 | int height, const SharpYuvOptions* options) { |
482 | 0 | const SharpYuvConversionMatrix* yuv_matrix = options->yuv_matrix; |
483 | 0 | SharpYuvTransferFunctionType transfer_type = options->transfer_type; |
484 | 0 | SharpYuvConversionMatrix scaled_matrix; |
485 | 0 | const int rgb_max = (1 << rgb_bit_depth) - 1; |
486 | 0 | const int rgb_round = 1 << (rgb_bit_depth - 1); |
487 | 0 | const int yuv_max = (1 << yuv_bit_depth) - 1; |
488 | 0 | const int sfix = GetPrecisionShift(rgb_bit_depth); |
489 | |
|
490 | 0 | if (width < 1 || height < 1 || width == INT_MAX || height == INT_MAX || |
491 | 0 | r_ptr == NULL || g_ptr == NULL || b_ptr == NULL || y_ptr == NULL || |
492 | 0 | u_ptr == NULL || v_ptr == NULL) { |
493 | 0 | return 0; |
494 | 0 | } |
495 | 0 | if (rgb_bit_depth != 8 && rgb_bit_depth != 10 && rgb_bit_depth != 12 && |
496 | 0 | rgb_bit_depth != 16) { |
497 | 0 | return 0; |
498 | 0 | } |
499 | 0 | if (yuv_bit_depth != 8 && yuv_bit_depth != 10 && yuv_bit_depth != 12) { |
500 | 0 | return 0; |
501 | 0 | } |
502 | 0 | if (rgb_bit_depth > 8 && (rgb_step % 2 != 0 || rgb_stride % 2 != 0)) { |
503 | | // Step/stride should be even for uint16_t buffers. |
504 | 0 | return 0; |
505 | 0 | } |
506 | 0 | { |
507 | 0 | const uint64_t yuv_bytes = (yuv_bit_depth > 8) ? 2 : 1; |
508 | 0 | const uint64_t uv_width = (width + 1) / 2; |
509 | 0 | const uint64_t abs_step = |
510 | 0 | (uint64_t)((rgb_step < 0) ? -(int64_t)rgb_step : (int64_t)rgb_step); |
511 | 0 | const uint64_t abs_stride = |
512 | 0 | (uint64_t)((rgb_stride < 0) ? -(int64_t)rgb_stride |
513 | 0 | : (int64_t)rgb_stride); |
514 | 0 | const uint64_t total_rgb_size = (uint64_t)height * abs_stride; |
515 | 0 | const uint64_t uv_height = (height + 1) / 2; |
516 | 0 | const uint64_t total_y_size = (uint64_t)height * y_stride; |
517 | 0 | const uint64_t total_u_size = uv_height * u_stride; |
518 | 0 | const uint64_t total_v_size = uv_height * v_stride; |
519 | |
|
520 | 0 | if (y_stride < 0 || (uint64_t)y_stride < (uint64_t)width * yuv_bytes || |
521 | 0 | u_stride < 0 || (uint64_t)u_stride < uv_width * yuv_bytes || |
522 | 0 | v_stride < 0 || (uint64_t)v_stride < uv_width * yuv_bytes) { |
523 | 0 | return 0; |
524 | 0 | } |
525 | 0 | if (abs_step == 0 || abs_stride < (uint64_t)width * abs_step) { |
526 | 0 | return 0; |
527 | 0 | } |
528 | 0 | if (total_rgb_size != (size_t)total_rgb_size || |
529 | 0 | total_y_size != (size_t)total_y_size || |
530 | 0 | total_u_size != (size_t)total_u_size || |
531 | 0 | total_v_size != (size_t)total_v_size) { |
532 | 0 | return 0; |
533 | 0 | } |
534 | 0 | } |
535 | 0 | if (yuv_bit_depth > 8 && |
536 | 0 | (y_stride % 2 != 0 || u_stride % 2 != 0 || v_stride % 2 != 0)) { |
537 | | // Stride should be even for uint16_t buffers. |
538 | 0 | return 0; |
539 | 0 | } |
540 | | // The address of the function pointer is used to avoid a read race. |
541 | 0 | SharpYuvInit((VP8CPUInfo)&SharpYuvGetCPUInfo); |
542 | | |
543 | | // Add scaling factor to go from rgb_bit_depth to yuv_bit_depth, to the |
544 | | // rgb->yuv conversion matrix. |
545 | 0 | if (rgb_bit_depth == yuv_bit_depth) { |
546 | 0 | memcpy(&scaled_matrix, yuv_matrix, sizeof(scaled_matrix)); |
547 | 0 | } else { |
548 | 0 | int i; |
549 | 0 | for (i = 0; i < 3; ++i) { |
550 | 0 | scaled_matrix.rgb_to_y[i] = |
551 | 0 | (yuv_matrix->rgb_to_y[i] * yuv_max + rgb_round) / rgb_max; |
552 | 0 | scaled_matrix.rgb_to_u[i] = |
553 | 0 | (yuv_matrix->rgb_to_u[i] * yuv_max + rgb_round) / rgb_max; |
554 | 0 | scaled_matrix.rgb_to_v[i] = |
555 | 0 | (yuv_matrix->rgb_to_v[i] * yuv_max + rgb_round) / rgb_max; |
556 | 0 | } |
557 | 0 | } |
558 | | // Also incorporate precision change scaling. |
559 | 0 | scaled_matrix.rgb_to_y[3] = Shift(yuv_matrix->rgb_to_y[3], sfix); |
560 | 0 | scaled_matrix.rgb_to_u[3] = Shift(yuv_matrix->rgb_to_u[3], sfix); |
561 | 0 | scaled_matrix.rgb_to_v[3] = Shift(yuv_matrix->rgb_to_v[3], sfix); |
562 | |
|
563 | 0 | return DoSharpArgbToYuv( |
564 | 0 | (const uint8_t*)r_ptr, (const uint8_t*)g_ptr, (const uint8_t*)b_ptr, |
565 | 0 | rgb_step, rgb_stride, rgb_bit_depth, (uint8_t*)y_ptr, y_stride, |
566 | 0 | (uint8_t*)u_ptr, u_stride, (uint8_t*)v_ptr, v_stride, yuv_bit_depth, |
567 | 0 | width, height, &scaled_matrix, transfer_type); |
568 | 0 | } |
569 | | |
570 | | //------------------------------------------------------------------------------ |