/src/libwebp/sharpyuv/sharpyuv_dsp.c
Line | Count | Source |
1 | | // Copyright 2022 Google Inc. All Rights Reserved. |
2 | | // |
3 | | // Use of this source code is governed by a BSD-style license |
4 | | // that can be found in the COPYING file in the root of the source |
5 | | // tree. An additional intellectual property rights grant can be found |
6 | | // in the file PATENTS. All contributing project authors may |
7 | | // be found in the AUTHORS file in the root of the source tree. |
8 | | // ----------------------------------------------------------------------------- |
9 | | // |
10 | | // Speed-critical functions for Sharp YUV. |
11 | | // |
12 | | // Author: Skal (pascal.massimino@gmail.com) |
13 | | |
14 | | #include "./sharpyuv_dsp.h" |
15 | | |
16 | | #include <assert.h> |
17 | | #include <stdlib.h> |
18 | | |
19 | | #include "./sharpyuv_cpu.h" |
20 | | #include "./sharpyuv_gamma.h" |
21 | | #include "src/dsp/cpu.h" |
22 | | #include "webp/types.h" |
23 | | |
24 | | //----------------------------------------------------------------------------- |
25 | | |
26 | | #if !WEBP_NEON_OMIT_C_CODE |
27 | | static uint64_t SharpYuvUpdateY_C(const uint16_t* ref, const uint16_t* src, |
28 | 0 | uint16_t* dst, int len, int bit_depth) { |
29 | 0 | uint64_t diff = 0; |
30 | 0 | int i; |
31 | 0 | const int max_y = (1 << bit_depth) - 1; |
32 | 0 | for (i = 0; i < len; ++i) { |
33 | 0 | const int diff_y = ref[i] - src[i]; |
34 | 0 | const int new_y = (int)dst[i] + diff_y; |
35 | 0 | dst[i] = SharpYuvClip16(new_y, max_y); |
36 | 0 | diff += (uint64_t)abs(diff_y); |
37 | 0 | } |
38 | 0 | return diff; |
39 | 0 | } |
40 | | |
41 | | static void SharpYuvUpdateRGB_C(const int16_t* ref, const int16_t* src, |
42 | 0 | int16_t* dst, int len) { |
43 | 0 | int i; |
44 | 0 | for (i = 0; i < len; ++i) { |
45 | 0 | const int diff_uv = ref[i] - src[i]; |
46 | 0 | dst[i] += diff_uv; |
47 | 0 | } |
48 | 0 | } |
49 | | |
50 | | static void SharpYuvFilterRow_C(const int16_t* A, const int16_t* B, int len, |
51 | | const uint16_t* best_y, uint16_t* out, |
52 | 0 | int bit_depth) { |
53 | 0 | int i; |
54 | 0 | const int max_y = (1 << bit_depth) - 1; |
55 | 0 | for (i = 0; i < len; ++i, ++A, ++B) { |
56 | 0 | const int v0 = (A[0] * 9 + A[1] * 3 + B[0] * 3 + B[1] + 8) >> 4; |
57 | 0 | const int v1 = (A[1] * 9 + A[0] * 3 + B[1] * 3 + B[0] + 8) >> 4; |
58 | 0 | out[2 * i + 0] = SharpYuvClip16(best_y[2 * i + 0] + v0, max_y); |
59 | 0 | out[2 * i + 1] = SharpYuvClip16(best_y[2 * i + 1] + v1, max_y); |
60 | 0 | } |
61 | 0 | } |
62 | | |
63 | | static void SharpYuvConvertRowY_C(const uint16_t* best_y, |
64 | | const int16_t* best_uv, int width, int uv_w, |
65 | | const int coeffs[4], int sfix, |
66 | 0 | int yuv_bit_depth, void* y_out) { |
67 | 0 | const int yuv_max = (1 << yuv_bit_depth) - 1; |
68 | 0 | int i; |
69 | 0 | for (i = 0; i < width; ++i) { |
70 | 0 | const int off = i >> 1; |
71 | 0 | const int W = best_y[i]; |
72 | 0 | const int r = best_uv[off + 0 * uv_w] + W; |
73 | 0 | const int g = best_uv[off + 1 * uv_w] + W; |
74 | 0 | const int b = best_uv[off + 2 * uv_w] + W; |
75 | 0 | const int y = SharpYuvConvertComponent(r, g, b, coeffs, sfix); |
76 | 0 | if (yuv_bit_depth <= 8) { |
77 | 0 | ((uint8_t*)y_out)[i] = (uint8_t)SharpYuvClip16(y, 255); |
78 | 0 | } else { |
79 | 0 | ((uint16_t*)y_out)[i] = SharpYuvClip16(y, yuv_max); |
80 | 0 | } |
81 | 0 | } |
82 | 0 | } |
83 | | |
84 | | static void SharpYuvConvertRowUV_C(const int16_t* best_uv, int uv_w, |
85 | | const int coeffs_u[4], const int coeffs_v[4], |
86 | | int sfix, int yuv_bit_depth, void* u_out, |
87 | 0 | void* v_out) { |
88 | 0 | const int yuv_max = (1 << yuv_bit_depth) - 1; |
89 | 0 | int i; |
90 | 0 | for (i = 0; i < uv_w; ++i) { |
91 | 0 | const int r = best_uv[i + 0 * uv_w]; |
92 | 0 | const int g = best_uv[i + 1 * uv_w]; |
93 | 0 | const int b = best_uv[i + 2 * uv_w]; |
94 | 0 | const int u = SharpYuvConvertComponent(r, g, b, coeffs_u, sfix); |
95 | 0 | const int v = SharpYuvConvertComponent(r, g, b, coeffs_v, sfix); |
96 | 0 | if (yuv_bit_depth <= 8) { |
97 | 0 | ((uint8_t*)u_out)[i] = (uint8_t)SharpYuvClip16(u, 255); |
98 | 0 | ((uint8_t*)v_out)[i] = (uint8_t)SharpYuvClip16(v, 255); |
99 | 0 | } else { |
100 | 0 | ((uint16_t*)u_out)[i] = SharpYuvClip16(u, yuv_max); |
101 | 0 | ((uint16_t*)v_out)[i] = SharpYuvClip16(v, yuv_max); |
102 | 0 | } |
103 | 0 | } |
104 | 0 | } |
105 | | |
106 | | // Fast path for the (default, most common) sRGB transfer function, used by |
107 | | // UpdateW/UpdateChroma in sharpyuv.c. |
108 | | static void SharpYuvUpdateWSrgb_C(const uint16_t* src, uint16_t* dst, int w, |
109 | 0 | int bit_depth) { |
110 | 0 | int i = 0; |
111 | 0 | do { |
112 | 0 | const uint32_t R = SharpYuvGammaToLinear(src[0 * w + i], bit_depth, |
113 | 0 | kSharpYuvTransferFunctionSrgb); |
114 | 0 | const uint32_t G = SharpYuvGammaToLinear(src[1 * w + i], bit_depth, |
115 | 0 | kSharpYuvTransferFunctionSrgb); |
116 | 0 | const uint32_t B = SharpYuvGammaToLinear(src[2 * w + i], bit_depth, |
117 | 0 | kSharpYuvTransferFunctionSrgb); |
118 | 0 | const int Y = SharpYuvRGBToGray(R, G, B); |
119 | 0 | dst[i] = SharpYuvLinearToGamma((uint32_t)Y, bit_depth, |
120 | 0 | kSharpYuvTransferFunctionSrgb); |
121 | 0 | } while (++i < w); |
122 | 0 | } |
123 | | |
124 | | static uint32_t ScaleDownSrgb_C(uint16_t a, uint16_t b, uint16_t c, uint16_t d, |
125 | 0 | int bit_depth) { |
126 | 0 | const uint32_t A = |
127 | 0 | SharpYuvGammaToLinear(a, bit_depth, kSharpYuvTransferFunctionSrgb); |
128 | 0 | const uint32_t B = |
129 | 0 | SharpYuvGammaToLinear(b, bit_depth, kSharpYuvTransferFunctionSrgb); |
130 | 0 | const uint32_t C = |
131 | 0 | SharpYuvGammaToLinear(c, bit_depth, kSharpYuvTransferFunctionSrgb); |
132 | 0 | const uint32_t D = |
133 | 0 | SharpYuvGammaToLinear(d, bit_depth, kSharpYuvTransferFunctionSrgb); |
134 | 0 | return SharpYuvLinearToGamma((A + B + C + D + 2) >> 2, bit_depth, |
135 | 0 | kSharpYuvTransferFunctionSrgb); |
136 | 0 | } |
137 | | |
138 | | static void SharpYuvUpdateChromaSrgb_C(const uint16_t* src1, |
139 | | const uint16_t* src2, int16_t* dst, |
140 | 0 | int uv_w, int bit_depth) { |
141 | 0 | int i = 0; |
142 | 0 | do { |
143 | 0 | const int r = |
144 | 0 | (int)ScaleDownSrgb_C(src1[0 * uv_w + 0], src1[0 * uv_w + 1], |
145 | 0 | src2[0 * uv_w + 0], src2[0 * uv_w + 1], bit_depth); |
146 | 0 | const int g = |
147 | 0 | (int)ScaleDownSrgb_C(src1[2 * uv_w + 0], src1[2 * uv_w + 1], |
148 | 0 | src2[2 * uv_w + 0], src2[2 * uv_w + 1], bit_depth); |
149 | 0 | const int b = |
150 | 0 | (int)ScaleDownSrgb_C(src1[4 * uv_w + 0], src1[4 * uv_w + 1], |
151 | 0 | src2[4 * uv_w + 0], src2[4 * uv_w + 1], bit_depth); |
152 | 0 | const int W = SharpYuvRGBToGray(r, g, b); |
153 | 0 | dst[0 * uv_w] = (int16_t)(r - W); |
154 | 0 | dst[1 * uv_w] = (int16_t)(g - W); |
155 | 0 | dst[2 * uv_w] = (int16_t)(b - W); |
156 | 0 | dst += 1; |
157 | 0 | src1 += 2; |
158 | 0 | src2 += 2; |
159 | 0 | } while (++i < uv_w); |
160 | 0 | } |
161 | | #endif // !WEBP_NEON_OMIT_C_CODE |
162 | | |
163 | | //----------------------------------------------------------------------------- |
164 | | |
165 | | uint64_t (*SharpYuvUpdateY)(const uint16_t* src, const uint16_t* ref, |
166 | | uint16_t* dst, int len, int bit_depth); |
167 | | void (*SharpYuvUpdateRGB)(const int16_t* src, const int16_t* ref, int16_t* dst, |
168 | | int len); |
169 | | void (*SharpYuvFilterRow)(const int16_t* A, const int16_t* B, int len, |
170 | | const uint16_t* best_y, uint16_t* out, int bit_depth); |
171 | | void (*SharpYuvConvertRowY)(const uint16_t* best_y, const int16_t* best_uv, |
172 | | int width, int uv_w, const int coeffs[4], int sfix, |
173 | | int yuv_bit_depth, void* y_out); |
174 | | void (*SharpYuvConvertRowUV)(const int16_t* best_uv, int uv_w, |
175 | | const int coeffs_u[4], const int coeffs_v[4], |
176 | | int sfix, int yuv_bit_depth, void* u_out, |
177 | | void* v_out); |
178 | | void (*SharpYuvUpdateWSrgb)(const uint16_t* src, uint16_t* dst, int w, |
179 | | int bit_depth); |
180 | | void (*SharpYuvUpdateChromaSrgb)(const uint16_t* src1, const uint16_t* src2, |
181 | | int16_t* dst, int uv_w, int bit_depth); |
182 | | |
183 | | extern VP8CPUInfo SharpYuvGetCPUInfo; |
184 | | extern void InitSharpYuvSSE2(void); |
185 | | extern void InitSharpYuvAVX2(void); |
186 | | extern void InitSharpYuvNEON(void); |
187 | | |
188 | 0 | void SharpYuvInitDsp(void) { |
189 | 0 | #if !WEBP_NEON_OMIT_C_CODE |
190 | 0 | SharpYuvUpdateY = SharpYuvUpdateY_C; |
191 | 0 | SharpYuvUpdateRGB = SharpYuvUpdateRGB_C; |
192 | 0 | SharpYuvFilterRow = SharpYuvFilterRow_C; |
193 | 0 | SharpYuvConvertRowY = SharpYuvConvertRowY_C; |
194 | 0 | SharpYuvConvertRowUV = SharpYuvConvertRowUV_C; |
195 | 0 | SharpYuvUpdateWSrgb = SharpYuvUpdateWSrgb_C; |
196 | 0 | SharpYuvUpdateChromaSrgb = SharpYuvUpdateChromaSrgb_C; |
197 | 0 | #endif |
198 | |
|
199 | 0 | if (SharpYuvGetCPUInfo != NULL) { |
200 | 0 | #if defined(WEBP_HAVE_SSE2) |
201 | 0 | if (SharpYuvGetCPUInfo(kSSE2)) { |
202 | 0 | InitSharpYuvSSE2(); |
203 | 0 | } |
204 | 0 | #endif // WEBP_HAVE_SSE2 |
205 | 0 | #if defined(WEBP_HAVE_AVX2) |
206 | 0 | if (SharpYuvGetCPUInfo(kAVX2)) { |
207 | 0 | InitSharpYuvAVX2(); |
208 | 0 | } |
209 | 0 | #endif // WEBP_HAVE_AVX2 |
210 | 0 | } |
211 | |
|
212 | | #if defined(WEBP_HAVE_NEON) |
213 | | if (WEBP_NEON_OMIT_C_CODE || |
214 | | (SharpYuvGetCPUInfo != NULL && SharpYuvGetCPUInfo(kNEON))) { |
215 | | InitSharpYuvNEON(); |
216 | | } |
217 | | #endif // WEBP_HAVE_NEON |
218 | |
|
219 | 0 | assert(SharpYuvUpdateY != NULL); |
220 | 0 | assert(SharpYuvUpdateRGB != NULL); |
221 | 0 | assert(SharpYuvFilterRow != NULL); |
222 | 0 | assert(SharpYuvConvertRowY != NULL); |
223 | 0 | assert(SharpYuvConvertRowUV != NULL); |
224 | 0 | assert(SharpYuvUpdateWSrgb != NULL); |
225 | 0 | assert(SharpYuvUpdateChromaSrgb != NULL); |
226 | 0 | } |