Coverage Report

Created: 2026-09-28 07:02

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/libwebp/sharpyuv/sharpyuv_avx2.c
Line
Count
Source
1
// Copyright 2026 Google Inc. All Rights Reserved.
2
//
3
// Use of this source code is governed by a BSD-style license
4
// that can be found in the COPYING file in the root of the source
5
// tree. An additional intellectual property rights grant can be found
6
// in the file PATENTS. All contributing project authors may
7
// be found in the AUTHORS file in the root of the source tree.
8
// -----------------------------------------------------------------------------
9
//
10
// AVX2 speed-critical functions for Sharp YUV.
11
12
#include "./sharpyuv_dsp.h"
13
#include "./sharpyuv_gamma.h"
14
15
#if defined(WEBP_USE_AVX2)
16
#include <assert.h>
17
#include <immintrin.h>
18
#include <stdlib.h>
19
20
#include "src/dsp/cpu.h"
21
#include "webp/types.h"
22
23
0
#define LOAD_256(P) _mm256_loadu_si256((const __m256i*)(P))
24
0
#define STORE_256(P, V) _mm256_storeu_si256((__m256i*)(P), (V))
25
0
#define LOAD_128(P) _mm_loadu_si128((const __m128i*)(P))
26
0
#define STORE_128(P, V) _mm_storeu_si128((__m128i*)(P), (V))
27
0
#define LOAD_64(P) _mm_loadl_epi64((const __m128i*)(P))
28
0
#define STORE_64(P, V) _mm_storel_epi64((__m128i*)(P), (V))
29
30
// Clamps 16x (resp. 8x) int16 lanes to [0, max].
31
0
static WEBP_INLINE __m256i Clip16_AVX2(__m256i v, __m256i max) {
32
0
  return _mm256_max_epi16(_mm256_min_epi16(v, max), _mm256_setzero_si256());
33
0
}
34
0
static WEBP_INLINE __m128i Clip16_128_AVX2(__m128i v, __m128i max) {
35
0
  return _mm_max_epi16(_mm_min_epi16(v, max), _mm_setzero_si128());
36
0
}
37
38
// Widens 8x int16 lanes into low/high 4x32 halves (sign-extending).
39
static WEBP_INLINE void WidenLoHi16To32_AVX2(__m128i v, __m128i* lo,
40
0
                                             __m128i* hi) {
41
0
  *lo = _mm_cvtepi16_epi32(v);
42
0
  *hi = _mm_cvtepi16_epi32(_mm_srli_si128(v, 8));
43
0
}
44
45
// e0 = (9*a0+3*a1+3*b0+b1+8)>>4, e1 symmetric. SUFFIX is epi16 or epi32.
46
#define SHARP_FILTER933_AVX2(SUFFIX, a0, a1, b0, b1, kCst8, e0_out, e1_out) \
47
0
  do {                                                                      \
48
0
    const __m256i a0b1 = _mm256_add_##SUFFIX(a0, b1);                       \
49
0
    const __m256i a1b0 = _mm256_add_##SUFFIX(a1, b0);                       \
50
0
    const __m256i a0a1b0b1 = _mm256_add_##SUFFIX(a0b1, a1b0);               \
51
0
    const __m256i a0a1b0b1_8 = _mm256_add_##SUFFIX(a0a1b0b1, kCst8);        \
52
0
    const __m256i a0b1_2 = _mm256_add_##SUFFIX(a0b1, a0b1);                 \
53
0
    const __m256i a1b0_2 = _mm256_add_##SUFFIX(a1b0, a1b0);                 \
54
0
    const __m256i c0_ =                                                     \
55
0
        _mm256_srai_##SUFFIX(_mm256_add_##SUFFIX(a0b1_2, a0a1b0b1_8), 3);   \
56
0
    const __m256i c1_ =                                                     \
57
0
        _mm256_srai_##SUFFIX(_mm256_add_##SUFFIX(a1b0_2, a0a1b0b1_8), 3);   \
58
0
    const __m256i d0_ = _mm256_add_##SUFFIX(c1_, a0);                       \
59
0
    const __m256i d1_ = _mm256_add_##SUFFIX(c0_, a1);                       \
60
0
    e0_out = _mm256_srai_##SUFFIX(d0_, 1);                                  \
61
0
    e1_out = _mm256_srai_##SUFFIX(d1_, 1);                                  \
62
0
  } while (0)
63
64
static uint64_t SharpYuvUpdateY_AVX2(const uint16_t* ref, const uint16_t* src,
65
0
                                     uint16_t* dst, int len, int bit_depth) {
66
0
  const int max_y = (1 << bit_depth) - 1;
67
0
  int i;
68
0
  const __m256i max = _mm256_set1_epi16(max_y);
69
0
  const __m256i one = _mm256_set1_epi16(1);
70
0
  __m256i sum = _mm256_setzero_si256();
71
0
  uint64_t diff;
72
73
0
  for (i = 0; i + 16 <= len; i += 16) {
74
0
    const __m256i A = LOAD_256(ref + i);
75
0
    const __m256i B = LOAD_256(src + i);
76
0
    const __m256i C = LOAD_256(dst + i);
77
0
    const __m256i D = _mm256_sub_epi16(A, B);        // diff_y
78
0
    const __m256i F = _mm256_add_epi16(C, D);        // new_y
79
0
    const __m256i AbsD = _mm256_abs_epi16(D);        // abs(diff_y)
80
0
    const __m256i I = _mm256_madd_epi16(AbsD, one);  // 32-bit pairs
81
0
    STORE_256(dst + i, Clip16_AVX2(F, max));
82
0
    sum = _mm256_add_epi32(sum, I);
83
0
  }
84
0
  {
85
0
    uint32_t tmp[8];
86
0
    STORE_256(tmp, sum);
87
0
    diff = (uint64_t)tmp[0] + tmp[1] + tmp[2] + tmp[3] + tmp[4] + tmp[5] +
88
0
           tmp[6] + tmp[7];
89
0
  }
90
0
  for (; i < len; ++i) {
91
0
    const int diff_y = ref[i] - src[i];
92
0
    const int new_y = (int)dst[i] + diff_y;
93
0
    dst[i] = SharpYuvClip16(new_y, max_y);
94
0
    diff += (uint64_t)abs(diff_y);
95
0
  }
96
0
  return diff;
97
0
}
98
99
static void SharpYuvUpdateRGB_AVX2(const int16_t* ref, const int16_t* src,
100
0
                                   int16_t* dst, int len) {
101
0
  int i;
102
0
  for (i = 0; i + 16 <= len; i += 16) {
103
0
    const __m256i A = LOAD_256(ref + i);
104
0
    const __m256i B = LOAD_256(src + i);
105
0
    const __m256i C = LOAD_256(dst + i);
106
0
    const __m256i D = _mm256_sub_epi16(A, B);  // diff_uv
107
0
    const __m256i E = _mm256_add_epi16(C, D);  // new_uv
108
0
    STORE_256(dst + i, E);
109
0
  }
110
0
  for (; i < len; ++i) {
111
0
    const int diff_uv = ref[i] - src[i];
112
0
    dst[i] += diff_uv;
113
0
  }
114
0
}
115
116
static void SharpYuvFilterRow16_AVX2(const int16_t* A, const int16_t* B,
117
                                     int len, const uint16_t* best_y,
118
0
                                     uint16_t* out, int bit_depth) {
119
0
  const int max_y = (1 << bit_depth) - 1;
120
0
  int i;
121
0
  const __m256i kCst8 = _mm256_set1_epi16(8);
122
0
  const __m256i max = _mm256_set1_epi16(max_y);
123
0
  for (i = 0; i + 16 <= len; i += 16) {
124
0
    const __m256i a0 = LOAD_256(A + i + 0);
125
0
    const __m256i a1 = LOAD_256(A + i + 1);
126
0
    const __m256i b0 = LOAD_256(B + i + 0);
127
0
    const __m256i b1 = LOAD_256(B + i + 1);
128
0
    __m256i e0, e1;
129
0
    SHARP_FILTER933_AVX2(epi16, a0, a1, b0, b1, kCst8, e0, e1);
130
0
    {
131
0
      const __m256i lo = _mm256_unpacklo_epi16(e0, e1);
132
0
      const __m256i hi = _mm256_unpackhi_epi16(e0, e1);
133
      // unpacklo/hi interleave within 128-bit lanes; permute to restore the
134
      // sequential (i, i+8) ordering matching best_y/out layout.
135
0
      const __m256i f0 = _mm256_permute2x128_si256(lo, hi, 0x20);
136
0
      const __m256i f1 = _mm256_permute2x128_si256(lo, hi, 0x31);
137
0
      const __m256i g0 = LOAD_256(best_y + 2 * i + 0);
138
0
      const __m256i g1 = LOAD_256(best_y + 2 * i + 16);
139
0
      STORE_256(out + 2 * i + 0, Clip16_AVX2(_mm256_add_epi16(g0, f0), max));
140
0
      STORE_256(out + 2 * i + 16, Clip16_AVX2(_mm256_add_epi16(g1, f1), max));
141
0
    }
142
0
  }
143
0
  for (; i < len; ++i) {
144
0
    const int v0 =
145
0
        (A[i + 0] * 9 + A[i + 1] * 3 + B[i + 0] * 3 + B[i + 1] + 8) >> 4;
146
0
    const int v1 =
147
0
        (A[i + 1] * 9 + A[i + 0] * 3 + B[i + 1] * 3 + B[i + 0] + 8) >> 4;
148
0
    out[2 * i + 0] = SharpYuvClip16(best_y[2 * i + 0] + v0, max_y);
149
0
    out[2 * i + 1] = SharpYuvClip16(best_y[2 * i + 1] + v1, max_y);
150
0
  }
151
0
}
152
153
// SharpYuvFilterRow16_AVX2's intermediate sums (e.g. a0a1b0b1_8) can exceed
154
// int16 range once bit_depth > 10 (chroma deltas get large enough that
155
// 2*(A+B) overflows signed 16 bits and wraps). This variant does the same
156
// computation in 32-bit lanes to stay safe, matching
157
// SharpYuvFilterRow32_SSE2/NEON.
158
static void SharpYuvFilterRow32_AVX2(const int16_t* A, const int16_t* B,
159
                                     int len, const uint16_t* best_y,
160
0
                                     uint16_t* out, int bit_depth) {
161
0
  const int max_y = (1 << bit_depth) - 1;
162
0
  int i;
163
0
  const __m256i kCst8 = _mm256_set1_epi32(8);
164
0
  const __m128i max = _mm_set1_epi16(max_y);
165
0
  for (i = 0; i + 8 <= len; i += 8) {
166
0
    const __m256i a0 = _mm256_cvtepi16_epi32(LOAD_128(A + i + 0));
167
0
    const __m256i a1 = _mm256_cvtepi16_epi32(LOAD_128(A + i + 1));
168
0
    const __m256i b0 = _mm256_cvtepi16_epi32(LOAD_128(B + i + 0));
169
0
    const __m256i b1 = _mm256_cvtepi16_epi32(LOAD_128(B + i + 1));
170
0
    __m256i e0, e1;
171
0
    SHARP_FILTER933_AVX2(epi32, a0, a1, b0, b1, kCst8, e0, e1);
172
    // Split into 128-bit halves and interleave/pack each independently (pure
173
    // SSE-style ops): avoids AVX2's cross-128-lane pack/unpack pitfalls.
174
0
    {
175
0
      const __m128i e0_lo = _mm256_castsi256_si128(e0);
176
0
      const __m128i e0_hi = _mm256_extracti128_si256(e0, 1);
177
0
      const __m128i e1_lo = _mm256_castsi256_si128(e1);
178
0
      const __m128i e1_hi = _mm256_extracti128_si256(e1, 1);
179
0
      const __m128i out_lo = _mm_packs_epi32(_mm_unpacklo_epi32(e0_lo, e1_lo),
180
0
                                             _mm_unpackhi_epi32(e0_lo, e1_lo));
181
0
      const __m128i out_hi = _mm_packs_epi32(_mm_unpacklo_epi32(e0_hi, e1_hi),
182
0
                                             _mm_unpackhi_epi32(e0_hi, e1_hi));
183
0
      const __m128i g_lo = LOAD_128(best_y + 2 * i + 0);
184
0
      const __m128i g_hi = LOAD_128(best_y + 2 * i + 8);
185
0
      STORE_128(out + 2 * i + 0,
186
0
                Clip16_128_AVX2(_mm_add_epi16(g_lo, out_lo), max));
187
0
      STORE_128(out + 2 * i + 8,
188
0
                Clip16_128_AVX2(_mm_add_epi16(g_hi, out_hi), max));
189
0
    }
190
0
  }
191
0
  for (; i < len; ++i) {
192
0
    const int a0b1 = A[i + 0] + B[i + 1];
193
0
    const int a1b0 = A[i + 1] + B[i + 0];
194
0
    const int a0a1b0b1 = a0b1 + a1b0 + 8;
195
0
    const int v0 = (8 * A[i + 0] + 2 * a1b0 + a0a1b0b1) >> 4;
196
0
    const int v1 = (8 * A[i + 1] + 2 * a0b1 + a0a1b0b1) >> 4;
197
0
    out[2 * i + 0] = SharpYuvClip16(best_y[2 * i + 0] + v0, max_y);
198
0
    out[2 * i + 1] = SharpYuvClip16(best_y[2 * i + 1] + v1, max_y);
199
0
  }
200
0
}
201
202
static void SharpYuvFilterRow_AVX2(const int16_t* A, const int16_t* B, int len,
203
                                   const uint16_t* best_y, uint16_t* out,
204
0
                                   int bit_depth) {
205
0
  if (bit_depth <= 10) {
206
0
    SharpYuvFilterRow16_AVX2(A, B, len, best_y, out, bit_depth);
207
0
  } else {
208
0
    SharpYuvFilterRow32_AVX2(A, B, len, best_y, out, bit_depth);
209
0
  }
210
0
}
211
212
//------------------------------------------------------------------------------
213
// Final W/RGB -> YUV reconstruction. Must stay numerically equivalent to
214
// RGBToYUVComponent() in sharpyuv.c, including its int64_t precision.
215
216
// Arithmetic right shift by a runtime amount n (1 <= n <= 63): AVX2 has no
217
// psraq, so build it from the logical shift plus a sign-extended top.
218
0
static WEBP_INLINE __m256i Sra64_AVX2(__m256i a, int n) {
219
0
  const __m128i cnt = _mm_cvtsi32_si128(n);
220
0
  const __m128i cnt2 = _mm_cvtsi32_si128(64 - n);
221
0
  const __m256i sign = _mm256_cmpgt_epi64(_mm256_setzero_si256(), a);
222
0
  const __m256i shifted = _mm256_srl_epi64(a, cnt);
223
0
  const __m256i topbits = _mm256_sll_epi64(sign, cnt2);
224
0
  return _mm256_or_si256(shifted, topbits);
225
0
}
226
227
// Truncating (non-saturating) narrow of 8x int32 lanes to 8x int16: matches
228
// a plain C (int16_t)/(uint16_t) cast, unlike _mm256_packs/packus_epi32.
229
0
static WEBP_INLINE __m128i TruncateNarrow32To16_AVX2(__m256i v) {
230
0
  const __m256i masked = _mm256_and_si256(v, _mm256_set1_epi32(0xffff));
231
0
  const __m256i packed = _mm256_packus_epi32(masked, masked);
232
0
  const __m128i lo = _mm256_castsi256_si128(packed);
233
0
  const __m128i hi = _mm256_extracti128_si256(packed, 1);
234
0
  return _mm_unpacklo_epi64(lo, hi);
235
0
}
236
237
// Narrow 4x64-bit lanes (2 groups of 2, from the mul_epi32 lane layout) to
238
// 4x32-bit, keeping only the low 32 bits of each 64-bit lane.
239
0
static WEBP_INLINE __m128i Narrow64To32_AVX2(__m256i v) {
240
0
  const __m256i shuffled = _mm256_shuffle_epi32(v, _MM_SHUFFLE(2, 0, 2, 0));
241
0
  const __m128i lo = _mm256_castsi256_si128(shuffled);
242
0
  const __m128i hi = _mm256_extracti128_si256(shuffled, 1);
243
0
  return _mm_unpacklo_epi64(lo, hi);
244
0
}
245
246
// round((R*coeffs[0] + G*coeffs[1] + B*coeffs[2] + bias) >> shift) for 4
247
// lanes, in exact 64-bit precision (matches the int64_t path in
248
// RGBToYUVComponent / RGBToGray).
249
static WEBP_INLINE __m128i Convert4_AVX2(__m128i R, __m128i G, __m128i B,
250
                                         const int coeffs[4], __m256i bias,
251
0
                                         int shift) {
252
0
  const __m256i R64 = _mm256_cvtepi32_epi64(R);
253
0
  const __m256i G64 = _mm256_cvtepi32_epi64(G);
254
0
  const __m256i B64 = _mm256_cvtepi32_epi64(B);
255
0
  __m256i acc = _mm256_add_epi64(
256
0
      bias, _mm256_mul_epi32(R64, _mm256_set1_epi64x(coeffs[0])));
257
0
  acc = _mm256_add_epi64(acc,
258
0
                         _mm256_mul_epi32(G64, _mm256_set1_epi64x(coeffs[1])));
259
0
  acc = _mm256_add_epi64(acc,
260
0
                         _mm256_mul_epi32(B64, _mm256_set1_epi64x(coeffs[2])));
261
0
  acc = Sra64_AVX2(acc, shift);
262
0
  return Narrow64To32_AVX2(acc);
263
0
}
264
265
static void SharpYuvConvertRowY_AVX2(const uint16_t* best_y,
266
                                     const int16_t* best_uv, int width,
267
                                     int uv_w, const int coeffs[4], int sfix,
268
0
                                     int yuv_bit_depth, void* y_out) {
269
0
  const int shift = SHARPYUV_YUV_FIX + sfix;
270
0
  const int64_t rounder = 1LL << (shift - 1);
271
0
  const __m256i bias = _mm256_set1_epi64x(coeffs[3] + rounder);
272
0
  const int yuv_max = (1 << yuv_bit_depth) - 1;
273
0
  int i = 0;
274
0
  for (; i + 8 <= width; i += 8) {
275
0
    const int off = i >> 1;
276
0
    const __m128i W = LOAD_128(best_y + i);
277
0
    const __m128i Ro = LOAD_64(best_uv + off + 0 * uv_w);
278
0
    const __m128i Go = LOAD_64(best_uv + off + 1 * uv_w);
279
0
    const __m128i Bo = LOAD_64(best_uv + off + 2 * uv_w);
280
0
    const __m128i Rv = _mm_add_epi16(_mm_unpacklo_epi16(Ro, Ro), W);
281
0
    const __m128i Gv = _mm_add_epi16(_mm_unpacklo_epi16(Go, Go), W);
282
0
    const __m128i Bv = _mm_add_epi16(_mm_unpacklo_epi16(Bo, Bo), W);
283
0
    __m128i R_lo, R_hi, G_lo, G_hi, B_lo, B_hi;
284
0
    WidenLoHi16To32_AVX2(Rv, &R_lo, &R_hi);
285
0
    WidenLoHi16To32_AVX2(Gv, &G_lo, &G_hi);
286
0
    WidenLoHi16To32_AVX2(Bv, &B_lo, &B_hi);
287
0
    {
288
0
      const __m128i y_lo = Convert4_AVX2(R_lo, G_lo, B_lo, coeffs, bias, shift);
289
0
      const __m128i y_hi = Convert4_AVX2(R_hi, G_hi, B_hi, coeffs, bias, shift);
290
0
      const __m128i y16 = _mm_packs_epi32(y_lo, y_hi);  // saturating:
291
                                                        // matches
292
                                                        // clip()/clip_8b().
293
0
      if (yuv_bit_depth <= 8) {
294
0
        STORE_64((uint8_t*)y_out + i, _mm_packus_epi16(y16, y16));
295
0
      } else {
296
0
        const __m128i maxv = _mm_set1_epi16((int16_t)yuv_max);
297
0
        STORE_128((uint16_t*)y_out + i, Clip16_128_AVX2(y16, maxv));
298
0
      }
299
0
    }
300
0
  }
301
0
  for (; i < width; ++i) {
302
0
    const int off = i >> 1;
303
0
    const int W = best_y[i];
304
0
    const int r = best_uv[off + 0 * uv_w] + W;
305
0
    const int g = best_uv[off + 1 * uv_w] + W;
306
0
    const int b = best_uv[off + 2 * uv_w] + W;
307
0
    const int y = SharpYuvConvertComponent(r, g, b, coeffs, sfix);
308
0
    if (yuv_bit_depth <= 8) {
309
0
      ((uint8_t*)y_out)[i] = (uint8_t)SharpYuvClip16(y, 255);
310
0
    } else {
311
0
      ((uint16_t*)y_out)[i] = SharpYuvClip16(y, yuv_max);
312
0
    }
313
0
  }
314
0
}
315
316
static void SharpYuvConvertRowUV_AVX2(const int16_t* best_uv, int uv_w,
317
                                      const int coeffs_u[4],
318
                                      const int coeffs_v[4], int sfix,
319
                                      int yuv_bit_depth, void* u_out,
320
0
                                      void* v_out) {
321
0
  const int shift = SHARPYUV_YUV_FIX + sfix;
322
0
  const int64_t rounder = 1LL << (shift - 1);
323
0
  const __m256i bias_u = _mm256_set1_epi64x(coeffs_u[3] + rounder);
324
0
  const __m256i bias_v = _mm256_set1_epi64x(coeffs_v[3] + rounder);
325
0
  const int yuv_max = (1 << yuv_bit_depth) - 1;
326
0
  int i = 0;
327
0
  for (; i + 8 <= uv_w; i += 8) {
328
0
    const __m128i R16 = LOAD_128(best_uv + i + 0 * uv_w);
329
0
    const __m128i G16 = LOAD_128(best_uv + i + 1 * uv_w);
330
0
    const __m128i B16 = LOAD_128(best_uv + i + 2 * uv_w);
331
0
    __m128i R_lo, R_hi, G_lo, G_hi, B_lo, B_hi;
332
0
    WidenLoHi16To32_AVX2(R16, &R_lo, &R_hi);
333
0
    WidenLoHi16To32_AVX2(G16, &G_lo, &G_hi);
334
0
    WidenLoHi16To32_AVX2(B16, &B_lo, &B_hi);
335
0
    {
336
0
      const __m128i u_lo =
337
0
          Convert4_AVX2(R_lo, G_lo, B_lo, coeffs_u, bias_u, shift);
338
0
      const __m128i u_hi =
339
0
          Convert4_AVX2(R_hi, G_hi, B_hi, coeffs_u, bias_u, shift);
340
0
      const __m128i v_lo =
341
0
          Convert4_AVX2(R_lo, G_lo, B_lo, coeffs_v, bias_v, shift);
342
0
      const __m128i v_hi =
343
0
          Convert4_AVX2(R_hi, G_hi, B_hi, coeffs_v, bias_v, shift);
344
0
      const __m128i u16 = _mm_packs_epi32(u_lo, u_hi);
345
0
      const __m128i v16 = _mm_packs_epi32(v_lo, v_hi);
346
0
      if (yuv_bit_depth <= 8) {
347
0
        STORE_64((uint8_t*)u_out + i, _mm_packus_epi16(u16, u16));
348
0
        STORE_64((uint8_t*)v_out + i, _mm_packus_epi16(v16, v16));
349
0
      } else {
350
0
        const __m128i maxv = _mm_set1_epi16((int16_t)yuv_max);
351
0
        STORE_128((uint16_t*)u_out + i, Clip16_128_AVX2(u16, maxv));
352
0
        STORE_128((uint16_t*)v_out + i, Clip16_128_AVX2(v16, maxv));
353
0
      }
354
0
    }
355
0
  }
356
0
  for (; i < uv_w; ++i) {
357
0
    const int r = best_uv[i + 0 * uv_w];
358
0
    const int g = best_uv[i + 1 * uv_w];
359
0
    const int b = best_uv[i + 2 * uv_w];
360
0
    const int u = SharpYuvConvertComponent(r, g, b, coeffs_u, sfix);
361
0
    const int v = SharpYuvConvertComponent(r, g, b, coeffs_v, sfix);
362
0
    if (yuv_bit_depth <= 8) {
363
0
      ((uint8_t*)u_out)[i] = (uint8_t)SharpYuvClip16(u, 255);
364
0
      ((uint8_t*)v_out)[i] = (uint8_t)SharpYuvClip16(v, 255);
365
0
    } else {
366
0
      ((uint16_t*)u_out)[i] = SharpYuvClip16(u, yuv_max);
367
0
      ((uint16_t*)v_out)[i] = SharpYuvClip16(v, yuv_max);
368
0
    }
369
0
  }
370
0
}
371
372
//------------------------------------------------------------------------------
373
// sRGB gamma<->linear fast paths for UpdateW/UpdateChroma.
374
// Leverage vpgatherdd gather instruction!
375
376
// Widens 8 samples to 16.16 fixed-point linear values using the
377
// sRGB gamma->linear table (shift_right = bit_depth - table_bits, >= 0).
378
0
static WEBP_INLINE __m256i GammaToLinear8_AVX2(__m128i v16, int shift_right) {
379
0
  const __m256i v32 = _mm256_cvtepu16_epi32(v16);
380
0
  const __m128i cnt = _mm_cvtsi32_si128(shift_right);
381
0
  const __m256i pos = _mm256_srl_epi32(v32, cnt);
382
0
  const __m256i x = _mm256_sub_epi32(v32, _mm256_sll_epi32(pos, cnt));
383
0
  const __m256i v0 =
384
0
      _mm256_i32gather_epi32((const int*)kSharpYuvGammaToLinearTabS, pos, 4);
385
0
  const __m256i v1 =
386
0
      _mm256_i32gather_epi32((const int*)kSharpYuvGammaToLinearTabS,
387
0
                             _mm256_add_epi32(pos, _mm256_set1_epi32(1)), 4);
388
  // d * x <= 65536 * 2^shift_right: fits in int32
389
0
  const __m256i prod = _mm256_mullo_epi32(_mm256_sub_epi32(v1, v0), x);
390
0
  const int half = (shift_right > 0) ? (1 << (shift_right - 1)) : 0;
391
0
  const __m256i sum = _mm256_add_epi32(prod, _mm256_set1_epi32(half));
392
0
  const __m256i shifted = _mm256_srl_epi32(sum, cnt);
393
0
  return _mm256_add_epi32(v0, shifted);
394
0
}
395
396
// Reverse direction: 8 16.16 fixed-point values -> 8 gamma samples.
397
0
static WEBP_INLINE __m256i LinearToGamma8_AVX2(__m256i v, int out_shift) {
398
0
  const __m256i pos = _mm256_srli_epi32(v, 7);  // table bits are fixed.
399
0
  const __m256i x = _mm256_sub_epi32(v, _mm256_slli_epi32(pos, 7));
400
0
  const __m256i v0_raw =
401
0
      _mm256_i32gather_epi32((const int*)kSharpYuvLinearToGammaTabS, pos, 4);
402
0
  const __m256i v1_raw =
403
0
      _mm256_i32gather_epi32((const int*)kSharpYuvLinearToGammaTabS,
404
0
                             _mm256_add_epi32(pos, _mm256_set1_epi32(1)), 4);
405
0
  const __m128i cnt = _mm_cvtsi32_si128(out_shift);
406
0
  const __m256i v0 = _mm256_srl_epi32(v0_raw, cnt);
407
0
  const __m256i v1 = _mm256_srl_epi32(v1_raw, cnt);
408
  // d * x <= 65536 * 128: fits in int32
409
0
  const __m256i prod = _mm256_mullo_epi32(_mm256_sub_epi32(v1, v0), x);
410
0
  const __m256i sum = _mm256_add_epi32(prod, _mm256_set1_epi32(64));
411
0
  const __m256i shifted = _mm256_srai_epi32(sum, 7);
412
0
  return _mm256_add_epi32(v0, shifted);
413
0
}
414
415
// (13933*R + 46871*G + 4732*B + 32768) >> 16, fits tightly in 32bits!
416
static WEBP_INLINE __m256i SharpYuvRGBToGray8_AVX2(__m256i R, __m256i G,
417
0
                                                   __m256i B) {
418
0
  static const int kGrayCoeffs[4] = {13933, 46871, 4732, 0};
419
0
  const __m256i bias = _mm256_set1_epi64x(1LL << (SHARPYUV_YUV_FIX - 1));
420
0
  const __m128i y_lo = Convert4_AVX2(
421
0
      _mm256_castsi256_si128(R), _mm256_castsi256_si128(G),
422
0
      _mm256_castsi256_si128(B), kGrayCoeffs, bias, SHARPYUV_YUV_FIX);
423
0
  const __m128i y_hi = Convert4_AVX2(
424
0
      _mm256_extracti128_si256(R, 1), _mm256_extracti128_si256(G, 1),
425
0
      _mm256_extracti128_si256(B, 1), kGrayCoeffs, bias, SHARPYUV_YUV_FIX);
426
0
  return _mm256_insertf128_si256(_mm256_castsi128_si256(y_lo), y_hi, 1);
427
0
}
428
429
static void SharpYuvUpdateWSrgb_AVX2(const uint16_t* src, uint16_t* dst, int w,
430
0
                                     int bit_depth) {
431
0
  const int shift_right = bit_depth - SHARPYUV_GAMMA_TO_LINEAR_TAB_BITS;
432
0
  const int out_shift = SHARPYUV_GAMMA_TO_LINEAR_BITS - bit_depth;
433
0
  int i = 0;
434
0
  assert(shift_right >= 0);
435
0
  for (; i + 8 <= w; i += 8) {
436
0
    const __m256i R =
437
0
        GammaToLinear8_AVX2(LOAD_128(src + 0 * w + i), shift_right);
438
0
    const __m256i G =
439
0
        GammaToLinear8_AVX2(LOAD_128(src + 1 * w + i), shift_right);
440
0
    const __m256i B =
441
0
        GammaToLinear8_AVX2(LOAD_128(src + 2 * w + i), shift_right);
442
0
    const __m256i Y = SharpYuvRGBToGray8_AVX2(R, G, B);
443
0
    STORE_128(dst + i,
444
0
              TruncateNarrow32To16_AVX2(LinearToGamma8_AVX2(Y, out_shift)));
445
0
  }
446
0
  for (; i < w; ++i) {
447
0
    const uint32_t R = SharpYuvGammaToLinear(src[0 * w + i], bit_depth,
448
0
                                             kSharpYuvTransferFunctionSrgb);
449
0
    const uint32_t G = SharpYuvGammaToLinear(src[1 * w + i], bit_depth,
450
0
                                             kSharpYuvTransferFunctionSrgb);
451
0
    const uint32_t B = SharpYuvGammaToLinear(src[2 * w + i], bit_depth,
452
0
                                             kSharpYuvTransferFunctionSrgb);
453
0
    const int Y = SharpYuvRGBToGray(R, G, B);
454
0
    dst[i] = SharpYuvLinearToGamma((uint32_t)Y, bit_depth,
455
0
                                   kSharpYuvTransferFunctionSrgb);
456
0
  }
457
0
}
458
459
// Deinterleaves 8 (a,b) pairs into a=[a0..a7], b=[b0..b7].
460
static WEBP_INLINE void Deinterleave8_AVX2(const uint16_t* p, __m128i* a,
461
0
                                           __m128i* b) {
462
0
  static const uint8_t kM[16] = {0, 1, 4, 5, 8,  9,  12, 13,
463
0
                                 2, 3, 6, 7, 10, 11, 14, 15};
464
0
  const __m256i mask = _mm256_broadcastsi128_si256(LOAD_128(kM));
465
0
  const __m256i v = LOAD_256(p);
466
0
  const __m256i shuffled = _mm256_shuffle_epi8(v, mask);
467
0
  const __m128i lane0 = _mm256_castsi256_si128(shuffled);
468
0
  const __m128i lane1 = _mm256_extracti128_si256(shuffled, 1);
469
0
  *a = _mm_unpacklo_epi64(lane0, lane1);
470
0
  *b = _mm_unpackhi_epi64(lane0, lane1);
471
0
}
472
473
// Box-downsamples 8 consecutive 2x2 uv-groups, matching ScaleDown().
474
static WEBP_INLINE __m128i ScaleDown8_AVX2(const uint16_t* p1,
475
                                           const uint16_t* p2, int shift_right,
476
0
                                           int out_shift) {
477
0
  __m128i a, b, c, d;
478
0
  Deinterleave8_AVX2(p1, &a, &b);
479
0
  Deinterleave8_AVX2(p2, &c, &d);
480
0
  {
481
0
    const __m256i A = GammaToLinear8_AVX2(a, shift_right);
482
0
    const __m256i B = GammaToLinear8_AVX2(b, shift_right);
483
0
    const __m256i C = GammaToLinear8_AVX2(c, shift_right);
484
0
    const __m256i D = GammaToLinear8_AVX2(d, shift_right);
485
0
    const __m256i sum =
486
0
        _mm256_add_epi32(_mm256_add_epi32(A, B), _mm256_add_epi32(C, D));
487
0
    const __m256i avg =
488
0
        _mm256_srli_epi32(_mm256_add_epi32(sum, _mm256_set1_epi32(2)), 2);
489
0
    return TruncateNarrow32To16_AVX2(LinearToGamma8_AVX2(avg, out_shift));
490
0
  }
491
0
}
492
493
static void SharpYuvUpdateChromaSrgb_AVX2(const uint16_t* src1,
494
                                          const uint16_t* src2, int16_t* dst,
495
0
                                          int uv_w, int bit_depth) {
496
0
  const int shift_right = bit_depth - SHARPYUV_GAMMA_TO_LINEAR_TAB_BITS;
497
0
  const int out_shift = SHARPYUV_GAMMA_TO_LINEAR_BITS - bit_depth;
498
0
  int i = 0;
499
0
  assert(shift_right >= 0);
500
0
  for (; i + 8 <= uv_w; i += 8) {
501
0
    const int off = 2 * i;
502
0
    const __m128i r16 = ScaleDown8_AVX2(
503
0
        src1 + off + 0 * uv_w, src2 + off + 0 * uv_w, shift_right, out_shift);
504
0
    const __m128i g16 = ScaleDown8_AVX2(
505
0
        src1 + off + 2 * uv_w, src2 + off + 2 * uv_w, shift_right, out_shift);
506
0
    const __m128i b16 = ScaleDown8_AVX2(
507
0
        src1 + off + 4 * uv_w, src2 + off + 4 * uv_w, shift_right, out_shift);
508
0
    const __m256i r32 = _mm256_cvtepu16_epi32(r16);
509
0
    const __m256i g32 = _mm256_cvtepu16_epi32(g16);
510
0
    const __m256i b32 = _mm256_cvtepu16_epi32(b16);
511
0
    const __m256i w32 = SharpYuvRGBToGray8_AVX2(r32, g32, b32);
512
0
    STORE_128(dst + i + 0 * uv_w,
513
0
              TruncateNarrow32To16_AVX2(_mm256_sub_epi32(r32, w32)));
514
0
    STORE_128(dst + i + 1 * uv_w,
515
0
              TruncateNarrow32To16_AVX2(_mm256_sub_epi32(g32, w32)));
516
0
    STORE_128(dst + i + 2 * uv_w,
517
0
              TruncateNarrow32To16_AVX2(_mm256_sub_epi32(b32, w32)));
518
0
  }
519
0
  for (; i < uv_w; ++i) {
520
0
    const int off = 2 * i;
521
0
    uint32_t rgb[3];
522
0
    int c;
523
0
    for (c = 0; c < 3; ++c) {
524
0
      const uint16_t* s1 = src1 + off + 2 * c * uv_w;
525
0
      const uint16_t* s2 = src2 + off + 2 * c * uv_w;
526
0
      const uint32_t A = SharpYuvGammaToLinear(s1[0], bit_depth,
527
0
                                               kSharpYuvTransferFunctionSrgb);
528
0
      const uint32_t B = SharpYuvGammaToLinear(s1[1], bit_depth,
529
0
                                               kSharpYuvTransferFunctionSrgb);
530
0
      const uint32_t C = SharpYuvGammaToLinear(s2[0], bit_depth,
531
0
                                               kSharpYuvTransferFunctionSrgb);
532
0
      const uint32_t D = SharpYuvGammaToLinear(s2[1], bit_depth,
533
0
                                               kSharpYuvTransferFunctionSrgb);
534
0
      rgb[c] = SharpYuvLinearToGamma((A + B + C + D + 2) >> 2, bit_depth,
535
0
                                     kSharpYuvTransferFunctionSrgb);
536
0
    }
537
0
    {
538
0
      const int W = SharpYuvRGBToGray(rgb[0], rgb[1], rgb[2]);
539
0
      dst[i + 0 * uv_w] = (int16_t)((int)rgb[0] - W);
540
0
      dst[i + 1 * uv_w] = (int16_t)((int)rgb[1] - W);
541
0
      dst[i + 2 * uv_w] = (int16_t)((int)rgb[2] - W);
542
0
    }
543
0
  }
544
0
}
545
546
//------------------------------------------------------------------------------
547
548
extern void InitSharpYuvAVX2(void);
549
550
0
WEBP_TSAN_IGNORE_FUNCTION void InitSharpYuvAVX2(void) {
551
0
  SharpYuvUpdateY = SharpYuvUpdateY_AVX2;
552
0
  SharpYuvUpdateRGB = SharpYuvUpdateRGB_AVX2;
553
0
  SharpYuvFilterRow = SharpYuvFilterRow_AVX2;
554
0
  SharpYuvConvertRowY = SharpYuvConvertRowY_AVX2;
555
0
  SharpYuvConvertRowUV = SharpYuvConvertRowUV_AVX2;
556
0
  SharpYuvUpdateWSrgb = SharpYuvUpdateWSrgb_AVX2;
557
0
  SharpYuvUpdateChromaSrgb = SharpYuvUpdateChromaSrgb_AVX2;
558
0
}
559
560
#else  // !WEBP_USE_AVX2
561
562
extern void InitSharpYuvAVX2(void);
563
564
void InitSharpYuvAVX2(void) {}
565
566
#endif  // WEBP_USE_AVX2