/src/libwebp/sharpyuv/sharpyuv_avx2.c
Line | Count | Source |
1 | | // Copyright 2026 Google Inc. All Rights Reserved. |
2 | | // |
3 | | // Use of this source code is governed by a BSD-style license |
4 | | // that can be found in the COPYING file in the root of the source |
5 | | // tree. An additional intellectual property rights grant can be found |
6 | | // in the file PATENTS. All contributing project authors may |
7 | | // be found in the AUTHORS file in the root of the source tree. |
8 | | // ----------------------------------------------------------------------------- |
9 | | // |
10 | | // AVX2 speed-critical functions for Sharp YUV. |
11 | | |
12 | | #include "./sharpyuv_dsp.h" |
13 | | #include "./sharpyuv_gamma.h" |
14 | | |
15 | | #if defined(WEBP_USE_AVX2) |
16 | | #include <assert.h> |
17 | | #include <immintrin.h> |
18 | | #include <stdlib.h> |
19 | | |
20 | | #include "src/dsp/cpu.h" |
21 | | #include "webp/types.h" |
22 | | |
23 | 0 | #define LOAD_256(P) _mm256_loadu_si256((const __m256i*)(P)) |
24 | 0 | #define STORE_256(P, V) _mm256_storeu_si256((__m256i*)(P), (V)) |
25 | 0 | #define LOAD_128(P) _mm_loadu_si128((const __m128i*)(P)) |
26 | 0 | #define STORE_128(P, V) _mm_storeu_si128((__m128i*)(P), (V)) |
27 | 0 | #define LOAD_64(P) _mm_loadl_epi64((const __m128i*)(P)) |
28 | 0 | #define STORE_64(P, V) _mm_storel_epi64((__m128i*)(P), (V)) |
29 | | |
30 | | // Clamps 16x (resp. 8x) int16 lanes to [0, max]. |
31 | 0 | static WEBP_INLINE __m256i Clip16_AVX2(__m256i v, __m256i max) { |
32 | 0 | return _mm256_max_epi16(_mm256_min_epi16(v, max), _mm256_setzero_si256()); |
33 | 0 | } |
34 | 0 | static WEBP_INLINE __m128i Clip16_128_AVX2(__m128i v, __m128i max) { |
35 | 0 | return _mm_max_epi16(_mm_min_epi16(v, max), _mm_setzero_si128()); |
36 | 0 | } |
37 | | |
38 | | // Widens 8x int16 lanes into low/high 4x32 halves (sign-extending). |
39 | | static WEBP_INLINE void WidenLoHi16To32_AVX2(__m128i v, __m128i* lo, |
40 | 0 | __m128i* hi) { |
41 | 0 | *lo = _mm_cvtepi16_epi32(v); |
42 | 0 | *hi = _mm_cvtepi16_epi32(_mm_srli_si128(v, 8)); |
43 | 0 | } |
44 | | |
45 | | // e0 = (9*a0+3*a1+3*b0+b1+8)>>4, e1 symmetric. SUFFIX is epi16 or epi32. |
46 | | #define SHARP_FILTER933_AVX2(SUFFIX, a0, a1, b0, b1, kCst8, e0_out, e1_out) \ |
47 | 0 | do { \ |
48 | 0 | const __m256i a0b1 = _mm256_add_##SUFFIX(a0, b1); \ |
49 | 0 | const __m256i a1b0 = _mm256_add_##SUFFIX(a1, b0); \ |
50 | 0 | const __m256i a0a1b0b1 = _mm256_add_##SUFFIX(a0b1, a1b0); \ |
51 | 0 | const __m256i a0a1b0b1_8 = _mm256_add_##SUFFIX(a0a1b0b1, kCst8); \ |
52 | 0 | const __m256i a0b1_2 = _mm256_add_##SUFFIX(a0b1, a0b1); \ |
53 | 0 | const __m256i a1b0_2 = _mm256_add_##SUFFIX(a1b0, a1b0); \ |
54 | 0 | const __m256i c0_ = \ |
55 | 0 | _mm256_srai_##SUFFIX(_mm256_add_##SUFFIX(a0b1_2, a0a1b0b1_8), 3); \ |
56 | 0 | const __m256i c1_ = \ |
57 | 0 | _mm256_srai_##SUFFIX(_mm256_add_##SUFFIX(a1b0_2, a0a1b0b1_8), 3); \ |
58 | 0 | const __m256i d0_ = _mm256_add_##SUFFIX(c1_, a0); \ |
59 | 0 | const __m256i d1_ = _mm256_add_##SUFFIX(c0_, a1); \ |
60 | 0 | e0_out = _mm256_srai_##SUFFIX(d0_, 1); \ |
61 | 0 | e1_out = _mm256_srai_##SUFFIX(d1_, 1); \ |
62 | 0 | } while (0) |
63 | | |
64 | | static uint64_t SharpYuvUpdateY_AVX2(const uint16_t* ref, const uint16_t* src, |
65 | 0 | uint16_t* dst, int len, int bit_depth) { |
66 | 0 | const int max_y = (1 << bit_depth) - 1; |
67 | 0 | int i; |
68 | 0 | const __m256i max = _mm256_set1_epi16(max_y); |
69 | 0 | const __m256i one = _mm256_set1_epi16(1); |
70 | 0 | __m256i sum = _mm256_setzero_si256(); |
71 | 0 | uint64_t diff; |
72 | |
|
73 | 0 | for (i = 0; i + 16 <= len; i += 16) { |
74 | 0 | const __m256i A = LOAD_256(ref + i); |
75 | 0 | const __m256i B = LOAD_256(src + i); |
76 | 0 | const __m256i C = LOAD_256(dst + i); |
77 | 0 | const __m256i D = _mm256_sub_epi16(A, B); // diff_y |
78 | 0 | const __m256i F = _mm256_add_epi16(C, D); // new_y |
79 | 0 | const __m256i AbsD = _mm256_abs_epi16(D); // abs(diff_y) |
80 | 0 | const __m256i I = _mm256_madd_epi16(AbsD, one); // 32-bit pairs |
81 | 0 | STORE_256(dst + i, Clip16_AVX2(F, max)); |
82 | 0 | sum = _mm256_add_epi32(sum, I); |
83 | 0 | } |
84 | 0 | { |
85 | 0 | uint32_t tmp[8]; |
86 | 0 | STORE_256(tmp, sum); |
87 | 0 | diff = (uint64_t)tmp[0] + tmp[1] + tmp[2] + tmp[3] + tmp[4] + tmp[5] + |
88 | 0 | tmp[6] + tmp[7]; |
89 | 0 | } |
90 | 0 | for (; i < len; ++i) { |
91 | 0 | const int diff_y = ref[i] - src[i]; |
92 | 0 | const int new_y = (int)dst[i] + diff_y; |
93 | 0 | dst[i] = SharpYuvClip16(new_y, max_y); |
94 | 0 | diff += (uint64_t)abs(diff_y); |
95 | 0 | } |
96 | 0 | return diff; |
97 | 0 | } |
98 | | |
99 | | static void SharpYuvUpdateRGB_AVX2(const int16_t* ref, const int16_t* src, |
100 | 0 | int16_t* dst, int len) { |
101 | 0 | int i; |
102 | 0 | for (i = 0; i + 16 <= len; i += 16) { |
103 | 0 | const __m256i A = LOAD_256(ref + i); |
104 | 0 | const __m256i B = LOAD_256(src + i); |
105 | 0 | const __m256i C = LOAD_256(dst + i); |
106 | 0 | const __m256i D = _mm256_sub_epi16(A, B); // diff_uv |
107 | 0 | const __m256i E = _mm256_add_epi16(C, D); // new_uv |
108 | 0 | STORE_256(dst + i, E); |
109 | 0 | } |
110 | 0 | for (; i < len; ++i) { |
111 | 0 | const int diff_uv = ref[i] - src[i]; |
112 | 0 | dst[i] += diff_uv; |
113 | 0 | } |
114 | 0 | } |
115 | | |
116 | | static void SharpYuvFilterRow16_AVX2(const int16_t* A, const int16_t* B, |
117 | | int len, const uint16_t* best_y, |
118 | 0 | uint16_t* out, int bit_depth) { |
119 | 0 | const int max_y = (1 << bit_depth) - 1; |
120 | 0 | int i; |
121 | 0 | const __m256i kCst8 = _mm256_set1_epi16(8); |
122 | 0 | const __m256i max = _mm256_set1_epi16(max_y); |
123 | 0 | for (i = 0; i + 16 <= len; i += 16) { |
124 | 0 | const __m256i a0 = LOAD_256(A + i + 0); |
125 | 0 | const __m256i a1 = LOAD_256(A + i + 1); |
126 | 0 | const __m256i b0 = LOAD_256(B + i + 0); |
127 | 0 | const __m256i b1 = LOAD_256(B + i + 1); |
128 | 0 | __m256i e0, e1; |
129 | 0 | SHARP_FILTER933_AVX2(epi16, a0, a1, b0, b1, kCst8, e0, e1); |
130 | 0 | { |
131 | 0 | const __m256i lo = _mm256_unpacklo_epi16(e0, e1); |
132 | 0 | const __m256i hi = _mm256_unpackhi_epi16(e0, e1); |
133 | | // unpacklo/hi interleave within 128-bit lanes; permute to restore the |
134 | | // sequential (i, i+8) ordering matching best_y/out layout. |
135 | 0 | const __m256i f0 = _mm256_permute2x128_si256(lo, hi, 0x20); |
136 | 0 | const __m256i f1 = _mm256_permute2x128_si256(lo, hi, 0x31); |
137 | 0 | const __m256i g0 = LOAD_256(best_y + 2 * i + 0); |
138 | 0 | const __m256i g1 = LOAD_256(best_y + 2 * i + 16); |
139 | 0 | STORE_256(out + 2 * i + 0, Clip16_AVX2(_mm256_add_epi16(g0, f0), max)); |
140 | 0 | STORE_256(out + 2 * i + 16, Clip16_AVX2(_mm256_add_epi16(g1, f1), max)); |
141 | 0 | } |
142 | 0 | } |
143 | 0 | for (; i < len; ++i) { |
144 | 0 | const int v0 = |
145 | 0 | (A[i + 0] * 9 + A[i + 1] * 3 + B[i + 0] * 3 + B[i + 1] + 8) >> 4; |
146 | 0 | const int v1 = |
147 | 0 | (A[i + 1] * 9 + A[i + 0] * 3 + B[i + 1] * 3 + B[i + 0] + 8) >> 4; |
148 | 0 | out[2 * i + 0] = SharpYuvClip16(best_y[2 * i + 0] + v0, max_y); |
149 | 0 | out[2 * i + 1] = SharpYuvClip16(best_y[2 * i + 1] + v1, max_y); |
150 | 0 | } |
151 | 0 | } |
152 | | |
153 | | // SharpYuvFilterRow16_AVX2's intermediate sums (e.g. a0a1b0b1_8) can exceed |
154 | | // int16 range once bit_depth > 10 (chroma deltas get large enough that |
155 | | // 2*(A+B) overflows signed 16 bits and wraps). This variant does the same |
156 | | // computation in 32-bit lanes to stay safe, matching |
157 | | // SharpYuvFilterRow32_SSE2/NEON. |
158 | | static void SharpYuvFilterRow32_AVX2(const int16_t* A, const int16_t* B, |
159 | | int len, const uint16_t* best_y, |
160 | 0 | uint16_t* out, int bit_depth) { |
161 | 0 | const int max_y = (1 << bit_depth) - 1; |
162 | 0 | int i; |
163 | 0 | const __m256i kCst8 = _mm256_set1_epi32(8); |
164 | 0 | const __m128i max = _mm_set1_epi16(max_y); |
165 | 0 | for (i = 0; i + 8 <= len; i += 8) { |
166 | 0 | const __m256i a0 = _mm256_cvtepi16_epi32(LOAD_128(A + i + 0)); |
167 | 0 | const __m256i a1 = _mm256_cvtepi16_epi32(LOAD_128(A + i + 1)); |
168 | 0 | const __m256i b0 = _mm256_cvtepi16_epi32(LOAD_128(B + i + 0)); |
169 | 0 | const __m256i b1 = _mm256_cvtepi16_epi32(LOAD_128(B + i + 1)); |
170 | 0 | __m256i e0, e1; |
171 | 0 | SHARP_FILTER933_AVX2(epi32, a0, a1, b0, b1, kCst8, e0, e1); |
172 | | // Split into 128-bit halves and interleave/pack each independently (pure |
173 | | // SSE-style ops): avoids AVX2's cross-128-lane pack/unpack pitfalls. |
174 | 0 | { |
175 | 0 | const __m128i e0_lo = _mm256_castsi256_si128(e0); |
176 | 0 | const __m128i e0_hi = _mm256_extracti128_si256(e0, 1); |
177 | 0 | const __m128i e1_lo = _mm256_castsi256_si128(e1); |
178 | 0 | const __m128i e1_hi = _mm256_extracti128_si256(e1, 1); |
179 | 0 | const __m128i out_lo = _mm_packs_epi32(_mm_unpacklo_epi32(e0_lo, e1_lo), |
180 | 0 | _mm_unpackhi_epi32(e0_lo, e1_lo)); |
181 | 0 | const __m128i out_hi = _mm_packs_epi32(_mm_unpacklo_epi32(e0_hi, e1_hi), |
182 | 0 | _mm_unpackhi_epi32(e0_hi, e1_hi)); |
183 | 0 | const __m128i g_lo = LOAD_128(best_y + 2 * i + 0); |
184 | 0 | const __m128i g_hi = LOAD_128(best_y + 2 * i + 8); |
185 | 0 | STORE_128(out + 2 * i + 0, |
186 | 0 | Clip16_128_AVX2(_mm_add_epi16(g_lo, out_lo), max)); |
187 | 0 | STORE_128(out + 2 * i + 8, |
188 | 0 | Clip16_128_AVX2(_mm_add_epi16(g_hi, out_hi), max)); |
189 | 0 | } |
190 | 0 | } |
191 | 0 | for (; i < len; ++i) { |
192 | 0 | const int a0b1 = A[i + 0] + B[i + 1]; |
193 | 0 | const int a1b0 = A[i + 1] + B[i + 0]; |
194 | 0 | const int a0a1b0b1 = a0b1 + a1b0 + 8; |
195 | 0 | const int v0 = (8 * A[i + 0] + 2 * a1b0 + a0a1b0b1) >> 4; |
196 | 0 | const int v1 = (8 * A[i + 1] + 2 * a0b1 + a0a1b0b1) >> 4; |
197 | 0 | out[2 * i + 0] = SharpYuvClip16(best_y[2 * i + 0] + v0, max_y); |
198 | 0 | out[2 * i + 1] = SharpYuvClip16(best_y[2 * i + 1] + v1, max_y); |
199 | 0 | } |
200 | 0 | } |
201 | | |
202 | | static void SharpYuvFilterRow_AVX2(const int16_t* A, const int16_t* B, int len, |
203 | | const uint16_t* best_y, uint16_t* out, |
204 | 0 | int bit_depth) { |
205 | 0 | if (bit_depth <= 10) { |
206 | 0 | SharpYuvFilterRow16_AVX2(A, B, len, best_y, out, bit_depth); |
207 | 0 | } else { |
208 | 0 | SharpYuvFilterRow32_AVX2(A, B, len, best_y, out, bit_depth); |
209 | 0 | } |
210 | 0 | } |
211 | | |
212 | | //------------------------------------------------------------------------------ |
213 | | // Final W/RGB -> YUV reconstruction. Must stay numerically equivalent to |
214 | | // RGBToYUVComponent() in sharpyuv.c, including its int64_t precision. |
215 | | |
216 | | // Arithmetic right shift by a runtime amount n (1 <= n <= 63): AVX2 has no |
217 | | // psraq, so build it from the logical shift plus a sign-extended top. |
218 | 0 | static WEBP_INLINE __m256i Sra64_AVX2(__m256i a, int n) { |
219 | 0 | const __m128i cnt = _mm_cvtsi32_si128(n); |
220 | 0 | const __m128i cnt2 = _mm_cvtsi32_si128(64 - n); |
221 | 0 | const __m256i sign = _mm256_cmpgt_epi64(_mm256_setzero_si256(), a); |
222 | 0 | const __m256i shifted = _mm256_srl_epi64(a, cnt); |
223 | 0 | const __m256i topbits = _mm256_sll_epi64(sign, cnt2); |
224 | 0 | return _mm256_or_si256(shifted, topbits); |
225 | 0 | } |
226 | | |
227 | | // Truncating (non-saturating) narrow of 8x int32 lanes to 8x int16: matches |
228 | | // a plain C (int16_t)/(uint16_t) cast, unlike _mm256_packs/packus_epi32. |
229 | 0 | static WEBP_INLINE __m128i TruncateNarrow32To16_AVX2(__m256i v) { |
230 | 0 | const __m256i masked = _mm256_and_si256(v, _mm256_set1_epi32(0xffff)); |
231 | 0 | const __m256i packed = _mm256_packus_epi32(masked, masked); |
232 | 0 | const __m128i lo = _mm256_castsi256_si128(packed); |
233 | 0 | const __m128i hi = _mm256_extracti128_si256(packed, 1); |
234 | 0 | return _mm_unpacklo_epi64(lo, hi); |
235 | 0 | } |
236 | | |
237 | | // Narrow 4x64-bit lanes (2 groups of 2, from the mul_epi32 lane layout) to |
238 | | // 4x32-bit, keeping only the low 32 bits of each 64-bit lane. |
239 | 0 | static WEBP_INLINE __m128i Narrow64To32_AVX2(__m256i v) { |
240 | 0 | const __m256i shuffled = _mm256_shuffle_epi32(v, _MM_SHUFFLE(2, 0, 2, 0)); |
241 | 0 | const __m128i lo = _mm256_castsi256_si128(shuffled); |
242 | 0 | const __m128i hi = _mm256_extracti128_si256(shuffled, 1); |
243 | 0 | return _mm_unpacklo_epi64(lo, hi); |
244 | 0 | } |
245 | | |
246 | | // round((R*coeffs[0] + G*coeffs[1] + B*coeffs[2] + bias) >> shift) for 4 |
247 | | // lanes, in exact 64-bit precision (matches the int64_t path in |
248 | | // RGBToYUVComponent / RGBToGray). |
249 | | static WEBP_INLINE __m128i Convert4_AVX2(__m128i R, __m128i G, __m128i B, |
250 | | const int coeffs[4], __m256i bias, |
251 | 0 | int shift) { |
252 | 0 | const __m256i R64 = _mm256_cvtepi32_epi64(R); |
253 | 0 | const __m256i G64 = _mm256_cvtepi32_epi64(G); |
254 | 0 | const __m256i B64 = _mm256_cvtepi32_epi64(B); |
255 | 0 | __m256i acc = _mm256_add_epi64( |
256 | 0 | bias, _mm256_mul_epi32(R64, _mm256_set1_epi64x(coeffs[0]))); |
257 | 0 | acc = _mm256_add_epi64(acc, |
258 | 0 | _mm256_mul_epi32(G64, _mm256_set1_epi64x(coeffs[1]))); |
259 | 0 | acc = _mm256_add_epi64(acc, |
260 | 0 | _mm256_mul_epi32(B64, _mm256_set1_epi64x(coeffs[2]))); |
261 | 0 | acc = Sra64_AVX2(acc, shift); |
262 | 0 | return Narrow64To32_AVX2(acc); |
263 | 0 | } |
264 | | |
265 | | static void SharpYuvConvertRowY_AVX2(const uint16_t* best_y, |
266 | | const int16_t* best_uv, int width, |
267 | | int uv_w, const int coeffs[4], int sfix, |
268 | 0 | int yuv_bit_depth, void* y_out) { |
269 | 0 | const int shift = SHARPYUV_YUV_FIX + sfix; |
270 | 0 | const int64_t rounder = 1LL << (shift - 1); |
271 | 0 | const __m256i bias = _mm256_set1_epi64x(coeffs[3] + rounder); |
272 | 0 | const int yuv_max = (1 << yuv_bit_depth) - 1; |
273 | 0 | int i = 0; |
274 | 0 | for (; i + 8 <= width; i += 8) { |
275 | 0 | const int off = i >> 1; |
276 | 0 | const __m128i W = LOAD_128(best_y + i); |
277 | 0 | const __m128i Ro = LOAD_64(best_uv + off + 0 * uv_w); |
278 | 0 | const __m128i Go = LOAD_64(best_uv + off + 1 * uv_w); |
279 | 0 | const __m128i Bo = LOAD_64(best_uv + off + 2 * uv_w); |
280 | 0 | const __m128i Rv = _mm_add_epi16(_mm_unpacklo_epi16(Ro, Ro), W); |
281 | 0 | const __m128i Gv = _mm_add_epi16(_mm_unpacklo_epi16(Go, Go), W); |
282 | 0 | const __m128i Bv = _mm_add_epi16(_mm_unpacklo_epi16(Bo, Bo), W); |
283 | 0 | __m128i R_lo, R_hi, G_lo, G_hi, B_lo, B_hi; |
284 | 0 | WidenLoHi16To32_AVX2(Rv, &R_lo, &R_hi); |
285 | 0 | WidenLoHi16To32_AVX2(Gv, &G_lo, &G_hi); |
286 | 0 | WidenLoHi16To32_AVX2(Bv, &B_lo, &B_hi); |
287 | 0 | { |
288 | 0 | const __m128i y_lo = Convert4_AVX2(R_lo, G_lo, B_lo, coeffs, bias, shift); |
289 | 0 | const __m128i y_hi = Convert4_AVX2(R_hi, G_hi, B_hi, coeffs, bias, shift); |
290 | 0 | const __m128i y16 = _mm_packs_epi32(y_lo, y_hi); // saturating: |
291 | | // matches |
292 | | // clip()/clip_8b(). |
293 | 0 | if (yuv_bit_depth <= 8) { |
294 | 0 | STORE_64((uint8_t*)y_out + i, _mm_packus_epi16(y16, y16)); |
295 | 0 | } else { |
296 | 0 | const __m128i maxv = _mm_set1_epi16((int16_t)yuv_max); |
297 | 0 | STORE_128((uint16_t*)y_out + i, Clip16_128_AVX2(y16, maxv)); |
298 | 0 | } |
299 | 0 | } |
300 | 0 | } |
301 | 0 | for (; i < width; ++i) { |
302 | 0 | const int off = i >> 1; |
303 | 0 | const int W = best_y[i]; |
304 | 0 | const int r = best_uv[off + 0 * uv_w] + W; |
305 | 0 | const int g = best_uv[off + 1 * uv_w] + W; |
306 | 0 | const int b = best_uv[off + 2 * uv_w] + W; |
307 | 0 | const int y = SharpYuvConvertComponent(r, g, b, coeffs, sfix); |
308 | 0 | if (yuv_bit_depth <= 8) { |
309 | 0 | ((uint8_t*)y_out)[i] = (uint8_t)SharpYuvClip16(y, 255); |
310 | 0 | } else { |
311 | 0 | ((uint16_t*)y_out)[i] = SharpYuvClip16(y, yuv_max); |
312 | 0 | } |
313 | 0 | } |
314 | 0 | } |
315 | | |
316 | | static void SharpYuvConvertRowUV_AVX2(const int16_t* best_uv, int uv_w, |
317 | | const int coeffs_u[4], |
318 | | const int coeffs_v[4], int sfix, |
319 | | int yuv_bit_depth, void* u_out, |
320 | 0 | void* v_out) { |
321 | 0 | const int shift = SHARPYUV_YUV_FIX + sfix; |
322 | 0 | const int64_t rounder = 1LL << (shift - 1); |
323 | 0 | const __m256i bias_u = _mm256_set1_epi64x(coeffs_u[3] + rounder); |
324 | 0 | const __m256i bias_v = _mm256_set1_epi64x(coeffs_v[3] + rounder); |
325 | 0 | const int yuv_max = (1 << yuv_bit_depth) - 1; |
326 | 0 | int i = 0; |
327 | 0 | for (; i + 8 <= uv_w; i += 8) { |
328 | 0 | const __m128i R16 = LOAD_128(best_uv + i + 0 * uv_w); |
329 | 0 | const __m128i G16 = LOAD_128(best_uv + i + 1 * uv_w); |
330 | 0 | const __m128i B16 = LOAD_128(best_uv + i + 2 * uv_w); |
331 | 0 | __m128i R_lo, R_hi, G_lo, G_hi, B_lo, B_hi; |
332 | 0 | WidenLoHi16To32_AVX2(R16, &R_lo, &R_hi); |
333 | 0 | WidenLoHi16To32_AVX2(G16, &G_lo, &G_hi); |
334 | 0 | WidenLoHi16To32_AVX2(B16, &B_lo, &B_hi); |
335 | 0 | { |
336 | 0 | const __m128i u_lo = |
337 | 0 | Convert4_AVX2(R_lo, G_lo, B_lo, coeffs_u, bias_u, shift); |
338 | 0 | const __m128i u_hi = |
339 | 0 | Convert4_AVX2(R_hi, G_hi, B_hi, coeffs_u, bias_u, shift); |
340 | 0 | const __m128i v_lo = |
341 | 0 | Convert4_AVX2(R_lo, G_lo, B_lo, coeffs_v, bias_v, shift); |
342 | 0 | const __m128i v_hi = |
343 | 0 | Convert4_AVX2(R_hi, G_hi, B_hi, coeffs_v, bias_v, shift); |
344 | 0 | const __m128i u16 = _mm_packs_epi32(u_lo, u_hi); |
345 | 0 | const __m128i v16 = _mm_packs_epi32(v_lo, v_hi); |
346 | 0 | if (yuv_bit_depth <= 8) { |
347 | 0 | STORE_64((uint8_t*)u_out + i, _mm_packus_epi16(u16, u16)); |
348 | 0 | STORE_64((uint8_t*)v_out + i, _mm_packus_epi16(v16, v16)); |
349 | 0 | } else { |
350 | 0 | const __m128i maxv = _mm_set1_epi16((int16_t)yuv_max); |
351 | 0 | STORE_128((uint16_t*)u_out + i, Clip16_128_AVX2(u16, maxv)); |
352 | 0 | STORE_128((uint16_t*)v_out + i, Clip16_128_AVX2(v16, maxv)); |
353 | 0 | } |
354 | 0 | } |
355 | 0 | } |
356 | 0 | for (; i < uv_w; ++i) { |
357 | 0 | const int r = best_uv[i + 0 * uv_w]; |
358 | 0 | const int g = best_uv[i + 1 * uv_w]; |
359 | 0 | const int b = best_uv[i + 2 * uv_w]; |
360 | 0 | const int u = SharpYuvConvertComponent(r, g, b, coeffs_u, sfix); |
361 | 0 | const int v = SharpYuvConvertComponent(r, g, b, coeffs_v, sfix); |
362 | 0 | if (yuv_bit_depth <= 8) { |
363 | 0 | ((uint8_t*)u_out)[i] = (uint8_t)SharpYuvClip16(u, 255); |
364 | 0 | ((uint8_t*)v_out)[i] = (uint8_t)SharpYuvClip16(v, 255); |
365 | 0 | } else { |
366 | 0 | ((uint16_t*)u_out)[i] = SharpYuvClip16(u, yuv_max); |
367 | 0 | ((uint16_t*)v_out)[i] = SharpYuvClip16(v, yuv_max); |
368 | 0 | } |
369 | 0 | } |
370 | 0 | } |
371 | | |
372 | | //------------------------------------------------------------------------------ |
373 | | // sRGB gamma<->linear fast paths for UpdateW/UpdateChroma. |
374 | | // Leverage vpgatherdd gather instruction! |
375 | | |
376 | | // Widens 8 samples to 16.16 fixed-point linear values using the |
377 | | // sRGB gamma->linear table (shift_right = bit_depth - table_bits, >= 0). |
378 | 0 | static WEBP_INLINE __m256i GammaToLinear8_AVX2(__m128i v16, int shift_right) { |
379 | 0 | const __m256i v32 = _mm256_cvtepu16_epi32(v16); |
380 | 0 | const __m128i cnt = _mm_cvtsi32_si128(shift_right); |
381 | 0 | const __m256i pos = _mm256_srl_epi32(v32, cnt); |
382 | 0 | const __m256i x = _mm256_sub_epi32(v32, _mm256_sll_epi32(pos, cnt)); |
383 | 0 | const __m256i v0 = |
384 | 0 | _mm256_i32gather_epi32((const int*)kSharpYuvGammaToLinearTabS, pos, 4); |
385 | 0 | const __m256i v1 = |
386 | 0 | _mm256_i32gather_epi32((const int*)kSharpYuvGammaToLinearTabS, |
387 | 0 | _mm256_add_epi32(pos, _mm256_set1_epi32(1)), 4); |
388 | | // d * x <= 65536 * 2^shift_right: fits in int32 |
389 | 0 | const __m256i prod = _mm256_mullo_epi32(_mm256_sub_epi32(v1, v0), x); |
390 | 0 | const int half = (shift_right > 0) ? (1 << (shift_right - 1)) : 0; |
391 | 0 | const __m256i sum = _mm256_add_epi32(prod, _mm256_set1_epi32(half)); |
392 | 0 | const __m256i shifted = _mm256_srl_epi32(sum, cnt); |
393 | 0 | return _mm256_add_epi32(v0, shifted); |
394 | 0 | } |
395 | | |
396 | | // Reverse direction: 8 16.16 fixed-point values -> 8 gamma samples. |
397 | 0 | static WEBP_INLINE __m256i LinearToGamma8_AVX2(__m256i v, int out_shift) { |
398 | 0 | const __m256i pos = _mm256_srli_epi32(v, 7); // table bits are fixed. |
399 | 0 | const __m256i x = _mm256_sub_epi32(v, _mm256_slli_epi32(pos, 7)); |
400 | 0 | const __m256i v0_raw = |
401 | 0 | _mm256_i32gather_epi32((const int*)kSharpYuvLinearToGammaTabS, pos, 4); |
402 | 0 | const __m256i v1_raw = |
403 | 0 | _mm256_i32gather_epi32((const int*)kSharpYuvLinearToGammaTabS, |
404 | 0 | _mm256_add_epi32(pos, _mm256_set1_epi32(1)), 4); |
405 | 0 | const __m128i cnt = _mm_cvtsi32_si128(out_shift); |
406 | 0 | const __m256i v0 = _mm256_srl_epi32(v0_raw, cnt); |
407 | 0 | const __m256i v1 = _mm256_srl_epi32(v1_raw, cnt); |
408 | | // d * x <= 65536 * 128: fits in int32 |
409 | 0 | const __m256i prod = _mm256_mullo_epi32(_mm256_sub_epi32(v1, v0), x); |
410 | 0 | const __m256i sum = _mm256_add_epi32(prod, _mm256_set1_epi32(64)); |
411 | 0 | const __m256i shifted = _mm256_srai_epi32(sum, 7); |
412 | 0 | return _mm256_add_epi32(v0, shifted); |
413 | 0 | } |
414 | | |
415 | | // (13933*R + 46871*G + 4732*B + 32768) >> 16, fits tightly in 32bits! |
416 | | static WEBP_INLINE __m256i SharpYuvRGBToGray8_AVX2(__m256i R, __m256i G, |
417 | 0 | __m256i B) { |
418 | 0 | static const int kGrayCoeffs[4] = {13933, 46871, 4732, 0}; |
419 | 0 | const __m256i bias = _mm256_set1_epi64x(1LL << (SHARPYUV_YUV_FIX - 1)); |
420 | 0 | const __m128i y_lo = Convert4_AVX2( |
421 | 0 | _mm256_castsi256_si128(R), _mm256_castsi256_si128(G), |
422 | 0 | _mm256_castsi256_si128(B), kGrayCoeffs, bias, SHARPYUV_YUV_FIX); |
423 | 0 | const __m128i y_hi = Convert4_AVX2( |
424 | 0 | _mm256_extracti128_si256(R, 1), _mm256_extracti128_si256(G, 1), |
425 | 0 | _mm256_extracti128_si256(B, 1), kGrayCoeffs, bias, SHARPYUV_YUV_FIX); |
426 | 0 | return _mm256_insertf128_si256(_mm256_castsi128_si256(y_lo), y_hi, 1); |
427 | 0 | } |
428 | | |
429 | | static void SharpYuvUpdateWSrgb_AVX2(const uint16_t* src, uint16_t* dst, int w, |
430 | 0 | int bit_depth) { |
431 | 0 | const int shift_right = bit_depth - SHARPYUV_GAMMA_TO_LINEAR_TAB_BITS; |
432 | 0 | const int out_shift = SHARPYUV_GAMMA_TO_LINEAR_BITS - bit_depth; |
433 | 0 | int i = 0; |
434 | 0 | assert(shift_right >= 0); |
435 | 0 | for (; i + 8 <= w; i += 8) { |
436 | 0 | const __m256i R = |
437 | 0 | GammaToLinear8_AVX2(LOAD_128(src + 0 * w + i), shift_right); |
438 | 0 | const __m256i G = |
439 | 0 | GammaToLinear8_AVX2(LOAD_128(src + 1 * w + i), shift_right); |
440 | 0 | const __m256i B = |
441 | 0 | GammaToLinear8_AVX2(LOAD_128(src + 2 * w + i), shift_right); |
442 | 0 | const __m256i Y = SharpYuvRGBToGray8_AVX2(R, G, B); |
443 | 0 | STORE_128(dst + i, |
444 | 0 | TruncateNarrow32To16_AVX2(LinearToGamma8_AVX2(Y, out_shift))); |
445 | 0 | } |
446 | 0 | for (; i < w; ++i) { |
447 | 0 | const uint32_t R = SharpYuvGammaToLinear(src[0 * w + i], bit_depth, |
448 | 0 | kSharpYuvTransferFunctionSrgb); |
449 | 0 | const uint32_t G = SharpYuvGammaToLinear(src[1 * w + i], bit_depth, |
450 | 0 | kSharpYuvTransferFunctionSrgb); |
451 | 0 | const uint32_t B = SharpYuvGammaToLinear(src[2 * w + i], bit_depth, |
452 | 0 | kSharpYuvTransferFunctionSrgb); |
453 | 0 | const int Y = SharpYuvRGBToGray(R, G, B); |
454 | 0 | dst[i] = SharpYuvLinearToGamma((uint32_t)Y, bit_depth, |
455 | 0 | kSharpYuvTransferFunctionSrgb); |
456 | 0 | } |
457 | 0 | } |
458 | | |
459 | | // Deinterleaves 8 (a,b) pairs into a=[a0..a7], b=[b0..b7]. |
460 | | static WEBP_INLINE void Deinterleave8_AVX2(const uint16_t* p, __m128i* a, |
461 | 0 | __m128i* b) { |
462 | 0 | static const uint8_t kM[16] = {0, 1, 4, 5, 8, 9, 12, 13, |
463 | 0 | 2, 3, 6, 7, 10, 11, 14, 15}; |
464 | 0 | const __m256i mask = _mm256_broadcastsi128_si256(LOAD_128(kM)); |
465 | 0 | const __m256i v = LOAD_256(p); |
466 | 0 | const __m256i shuffled = _mm256_shuffle_epi8(v, mask); |
467 | 0 | const __m128i lane0 = _mm256_castsi256_si128(shuffled); |
468 | 0 | const __m128i lane1 = _mm256_extracti128_si256(shuffled, 1); |
469 | 0 | *a = _mm_unpacklo_epi64(lane0, lane1); |
470 | 0 | *b = _mm_unpackhi_epi64(lane0, lane1); |
471 | 0 | } |
472 | | |
473 | | // Box-downsamples 8 consecutive 2x2 uv-groups, matching ScaleDown(). |
474 | | static WEBP_INLINE __m128i ScaleDown8_AVX2(const uint16_t* p1, |
475 | | const uint16_t* p2, int shift_right, |
476 | 0 | int out_shift) { |
477 | 0 | __m128i a, b, c, d; |
478 | 0 | Deinterleave8_AVX2(p1, &a, &b); |
479 | 0 | Deinterleave8_AVX2(p2, &c, &d); |
480 | 0 | { |
481 | 0 | const __m256i A = GammaToLinear8_AVX2(a, shift_right); |
482 | 0 | const __m256i B = GammaToLinear8_AVX2(b, shift_right); |
483 | 0 | const __m256i C = GammaToLinear8_AVX2(c, shift_right); |
484 | 0 | const __m256i D = GammaToLinear8_AVX2(d, shift_right); |
485 | 0 | const __m256i sum = |
486 | 0 | _mm256_add_epi32(_mm256_add_epi32(A, B), _mm256_add_epi32(C, D)); |
487 | 0 | const __m256i avg = |
488 | 0 | _mm256_srli_epi32(_mm256_add_epi32(sum, _mm256_set1_epi32(2)), 2); |
489 | 0 | return TruncateNarrow32To16_AVX2(LinearToGamma8_AVX2(avg, out_shift)); |
490 | 0 | } |
491 | 0 | } |
492 | | |
493 | | static void SharpYuvUpdateChromaSrgb_AVX2(const uint16_t* src1, |
494 | | const uint16_t* src2, int16_t* dst, |
495 | 0 | int uv_w, int bit_depth) { |
496 | 0 | const int shift_right = bit_depth - SHARPYUV_GAMMA_TO_LINEAR_TAB_BITS; |
497 | 0 | const int out_shift = SHARPYUV_GAMMA_TO_LINEAR_BITS - bit_depth; |
498 | 0 | int i = 0; |
499 | 0 | assert(shift_right >= 0); |
500 | 0 | for (; i + 8 <= uv_w; i += 8) { |
501 | 0 | const int off = 2 * i; |
502 | 0 | const __m128i r16 = ScaleDown8_AVX2( |
503 | 0 | src1 + off + 0 * uv_w, src2 + off + 0 * uv_w, shift_right, out_shift); |
504 | 0 | const __m128i g16 = ScaleDown8_AVX2( |
505 | 0 | src1 + off + 2 * uv_w, src2 + off + 2 * uv_w, shift_right, out_shift); |
506 | 0 | const __m128i b16 = ScaleDown8_AVX2( |
507 | 0 | src1 + off + 4 * uv_w, src2 + off + 4 * uv_w, shift_right, out_shift); |
508 | 0 | const __m256i r32 = _mm256_cvtepu16_epi32(r16); |
509 | 0 | const __m256i g32 = _mm256_cvtepu16_epi32(g16); |
510 | 0 | const __m256i b32 = _mm256_cvtepu16_epi32(b16); |
511 | 0 | const __m256i w32 = SharpYuvRGBToGray8_AVX2(r32, g32, b32); |
512 | 0 | STORE_128(dst + i + 0 * uv_w, |
513 | 0 | TruncateNarrow32To16_AVX2(_mm256_sub_epi32(r32, w32))); |
514 | 0 | STORE_128(dst + i + 1 * uv_w, |
515 | 0 | TruncateNarrow32To16_AVX2(_mm256_sub_epi32(g32, w32))); |
516 | 0 | STORE_128(dst + i + 2 * uv_w, |
517 | 0 | TruncateNarrow32To16_AVX2(_mm256_sub_epi32(b32, w32))); |
518 | 0 | } |
519 | 0 | for (; i < uv_w; ++i) { |
520 | 0 | const int off = 2 * i; |
521 | 0 | uint32_t rgb[3]; |
522 | 0 | int c; |
523 | 0 | for (c = 0; c < 3; ++c) { |
524 | 0 | const uint16_t* s1 = src1 + off + 2 * c * uv_w; |
525 | 0 | const uint16_t* s2 = src2 + off + 2 * c * uv_w; |
526 | 0 | const uint32_t A = SharpYuvGammaToLinear(s1[0], bit_depth, |
527 | 0 | kSharpYuvTransferFunctionSrgb); |
528 | 0 | const uint32_t B = SharpYuvGammaToLinear(s1[1], bit_depth, |
529 | 0 | kSharpYuvTransferFunctionSrgb); |
530 | 0 | const uint32_t C = SharpYuvGammaToLinear(s2[0], bit_depth, |
531 | 0 | kSharpYuvTransferFunctionSrgb); |
532 | 0 | const uint32_t D = SharpYuvGammaToLinear(s2[1], bit_depth, |
533 | 0 | kSharpYuvTransferFunctionSrgb); |
534 | 0 | rgb[c] = SharpYuvLinearToGamma((A + B + C + D + 2) >> 2, bit_depth, |
535 | 0 | kSharpYuvTransferFunctionSrgb); |
536 | 0 | } |
537 | 0 | { |
538 | 0 | const int W = SharpYuvRGBToGray(rgb[0], rgb[1], rgb[2]); |
539 | 0 | dst[i + 0 * uv_w] = (int16_t)((int)rgb[0] - W); |
540 | 0 | dst[i + 1 * uv_w] = (int16_t)((int)rgb[1] - W); |
541 | 0 | dst[i + 2 * uv_w] = (int16_t)((int)rgb[2] - W); |
542 | 0 | } |
543 | 0 | } |
544 | 0 | } |
545 | | |
546 | | //------------------------------------------------------------------------------ |
547 | | |
548 | | extern void InitSharpYuvAVX2(void); |
549 | | |
550 | 0 | WEBP_TSAN_IGNORE_FUNCTION void InitSharpYuvAVX2(void) { |
551 | 0 | SharpYuvUpdateY = SharpYuvUpdateY_AVX2; |
552 | 0 | SharpYuvUpdateRGB = SharpYuvUpdateRGB_AVX2; |
553 | 0 | SharpYuvFilterRow = SharpYuvFilterRow_AVX2; |
554 | 0 | SharpYuvConvertRowY = SharpYuvConvertRowY_AVX2; |
555 | 0 | SharpYuvConvertRowUV = SharpYuvConvertRowUV_AVX2; |
556 | 0 | SharpYuvUpdateWSrgb = SharpYuvUpdateWSrgb_AVX2; |
557 | 0 | SharpYuvUpdateChromaSrgb = SharpYuvUpdateChromaSrgb_AVX2; |
558 | 0 | } |
559 | | |
560 | | #else // !WEBP_USE_AVX2 |
561 | | |
562 | | extern void InitSharpYuvAVX2(void); |
563 | | |
564 | | void InitSharpYuvAVX2(void) {} |
565 | | |
566 | | #endif // WEBP_USE_AVX2 |