Coverage Report

Created: 2026-09-28 06:47

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/libwebp/src/dsp/lossless_enc_avx2.c
Line
Count
Source
1
// Copyright 2025 Google Inc. All Rights Reserved.
2
//
3
// Use of this source code is governed by a BSD-style license
4
// that can be found in the COPYING file in the root of the source
5
// tree. An additional intellectual property rights grant can be found
6
// in the file PATENTS. All contributing project authors may
7
// be found in the AUTHORS file in the root of the source tree.
8
// -----------------------------------------------------------------------------
9
//
10
// AVX2 variant of methods for lossless encoder
11
//
12
// Author: Vincent Rabaud (vrabaud@google.com)
13
14
#include "src/dsp/dsp.h"
15
16
#if defined(WEBP_USE_AVX2)
17
#include <assert.h>
18
#include <emmintrin.h>
19
#include <immintrin.h>
20
#include <stddef.h>
21
22
#include "src/dsp/cpu.h"
23
#include "src/dsp/lossless.h"
24
#include "src/dsp/lossless_common.h"
25
#include "src/utils/utils.h"
26
#include "src/webp/format_constants.h"
27
#include "src/webp/types.h"
28
29
//------------------------------------------------------------------------------
30
// Subtract-Green Transform
31
32
static void SubtractGreenFromBlueAndRed_AVX2(uint32_t* argb_data,
33
0
                                             int num_pixels) {
34
0
  int i;
35
0
  const __m256i kCstShuffle = _mm256_set_epi8(
36
0
      -1, 29, -1, 29, -1, 25, -1, 25, -1, 21, -1, 21, -1, 17, -1, 17, -1, 13,
37
0
      -1, 13, -1, 9, -1, 9, -1, 5, -1, 5, -1, 1, -1, 1);
38
0
  for (i = 0; i + 8 <= num_pixels; i += 8) {
39
0
    const __m256i in = _mm256_loadu_si256((__m256i*)&argb_data[i]);  // argb
40
0
    const __m256i in_0g0g = _mm256_shuffle_epi8(in, kCstShuffle);
41
0
    const __m256i out = _mm256_sub_epi8(in, in_0g0g);
42
0
    _mm256_storeu_si256((__m256i*)&argb_data[i], out);
43
0
  }
44
  // fallthrough and finish off with plain-SSE
45
0
  if (i != num_pixels) {
46
0
    VP8LSubtractGreenFromBlueAndRed_SSE(argb_data + i, num_pixels - i);
47
0
  }
48
0
}
49
50
//------------------------------------------------------------------------------
51
// Color Transform
52
53
// For sign-extended multiplying constants, pre-shifted by 5:
54
#define CST_5b(X) (((int16_t)((uint16_t)(X) << 8)) >> 5)
55
56
#define MK_CST_16(HI, LO) \
57
0
  _mm256_set1_epi32((int)(((uint32_t)(HI) << 16) | ((LO) & 0xffff)))
58
59
static void TransformColor_AVX2(const VP8LMultipliers* WEBP_RESTRICT const m,
60
                                uint32_t* WEBP_RESTRICT argb_data,
61
0
                                int num_pixels) {
62
0
  const __m256i mults_rb =
63
0
      MK_CST_16(CST_5b(m->green_to_red), CST_5b(m->green_to_blue));
64
0
  const __m256i mults_b2 = MK_CST_16(CST_5b(m->red_to_blue), 0);
65
0
  const __m256i mask_rb = _mm256_set1_epi32(0x00ff00ff);  // red-blue masks
66
0
  const __m256i kCstShuffle = _mm256_set_epi8(
67
0
      29, -1, 29, -1, 25, -1, 25, -1, 21, -1, 21, -1, 17, -1, 17, -1, 13, -1,
68
0
      13, -1, 9, -1, 9, -1, 5, -1, 5, -1, 1, -1, 1, -1);
69
0
  int i;
70
0
  for (i = 0; i + 8 <= num_pixels; i += 8) {
71
0
    const __m256i in = _mm256_loadu_si256((__m256i*)&argb_data[i]);  // argb
72
0
    const __m256i A = _mm256_shuffle_epi8(in, kCstShuffle);          // g0g0
73
0
    const __m256i B = _mm256_mulhi_epi16(A, mults_rb);  // x dr  x db1
74
0
    const __m256i C = _mm256_slli_epi16(in, 8);         // r 0   b   0
75
0
    const __m256i D = _mm256_mulhi_epi16(C, mults_b2);  // x db2 0   0
76
0
    const __m256i E = _mm256_srli_epi32(D, 16);         // 0 0   x db2
77
0
    const __m256i F = _mm256_add_epi8(E, B);            // x dr  x  db
78
0
    const __m256i G = _mm256_and_si256(F, mask_rb);     // 0 dr  0  db
79
0
    const __m256i out = _mm256_sub_epi8(in, G);
80
0
    _mm256_storeu_si256((__m256i*)&argb_data[i], out);
81
0
  }
82
  // fallthrough and finish off with plain-C
83
0
  if (i != num_pixels) {
84
0
    VP8LTransformColor_SSE(m, argb_data + i, num_pixels - i);
85
0
  }
86
0
}
87
88
//------------------------------------------------------------------------------
89
#define SPAN 16
90
static void CollectColorBlueTransforms_AVX2(const uint32_t* WEBP_RESTRICT argb,
91
                                            int stride, int tile_width,
92
                                            int tile_height, int green_to_blue,
93
0
                                            int red_to_blue, uint32_t histo[]) {
94
0
  const __m256i mult =
95
0
      MK_CST_16(CST_5b(red_to_blue) + 256, CST_5b(green_to_blue));
96
0
  const __m256i perm = _mm256_setr_epi8(
97
0
      -1, 1, -1, 2, -1, 5, -1, 6, -1, 9, -1, 10, -1, 13, -1, 14, -1, 17, -1, 18,
98
0
      -1, 21, -1, 22, -1, 25, -1, 26, -1, 29, -1, 30);
99
0
  if (tile_width >= 8) {
100
0
    int y, i;
101
0
    for (y = 0; y < tile_height; ++y) {
102
0
      uint8_t values[32];
103
0
      const uint32_t* const src = argb + y * stride;
104
0
      const __m256i A1 = _mm256_loadu_si256((const __m256i*)src);
105
0
      const __m256i B1 = _mm256_shuffle_epi8(A1, perm);
106
0
      const __m256i C1 = _mm256_mulhi_epi16(B1, mult);
107
0
      const __m256i D1 = _mm256_sub_epi16(A1, C1);
108
0
      __m256i E = _mm256_add_epi16(_mm256_srli_epi32(D1, 16), D1);
109
0
      int x;
110
0
      for (x = 8; x + 8 <= tile_width; x += 8) {
111
0
        const __m256i A2 = _mm256_loadu_si256((const __m256i*)(src + x));
112
0
        __m256i B2, C2, D2;
113
0
        _mm256_storeu_si256((__m256i*)values, E);
114
0
        for (i = 0; i < 32; i += 4) ++histo[values[i]];
115
0
        B2 = _mm256_shuffle_epi8(A2, perm);
116
0
        C2 = _mm256_mulhi_epi16(B2, mult);
117
0
        D2 = _mm256_sub_epi16(A2, C2);
118
0
        E = _mm256_add_epi16(_mm256_srli_epi32(D2, 16), D2);
119
0
      }
120
0
      _mm256_storeu_si256((__m256i*)values, E);
121
0
      for (i = 0; i < 32; i += 4) ++histo[values[i]];
122
0
    }
123
0
  }
124
0
  {
125
0
    const int left_over = tile_width & 7;
126
0
    if (left_over > 0) {
127
0
      VP8LCollectColorBlueTransforms_SSE(argb + tile_width - left_over, stride,
128
0
                                         left_over, tile_height, green_to_blue,
129
0
                                         red_to_blue, histo);
130
0
    }
131
0
  }
132
0
}
133
134
static void CollectColorRedTransforms_AVX2(const uint32_t* WEBP_RESTRICT argb,
135
                                           int stride, int tile_width,
136
                                           int tile_height, int green_to_red,
137
0
                                           uint32_t histo[]) {
138
0
  const __m256i mult = MK_CST_16(0, CST_5b(green_to_red));
139
0
  const __m256i mask_g = _mm256_set1_epi32(0x0000ff00);
140
0
  if (tile_width >= 8) {
141
0
    int y, i;
142
0
    for (y = 0; y < tile_height; ++y) {
143
0
      uint8_t values[32];
144
0
      const uint32_t* const src = argb + y * stride;
145
0
      const __m256i A1 = _mm256_loadu_si256((const __m256i*)src);
146
0
      const __m256i B1 = _mm256_and_si256(A1, mask_g);
147
0
      const __m256i C1 = _mm256_madd_epi16(B1, mult);
148
0
      __m256i D = _mm256_sub_epi16(A1, C1);
149
0
      int x;
150
0
      for (x = 8; x + 8 <= tile_width; x += 8) {
151
0
        const __m256i A2 = _mm256_loadu_si256((const __m256i*)(src + x));
152
0
        __m256i B2, C2;
153
0
        _mm256_storeu_si256((__m256i*)values, D);
154
0
        for (i = 2; i < 32; i += 4) ++histo[values[i]];
155
0
        B2 = _mm256_and_si256(A2, mask_g);
156
0
        C2 = _mm256_madd_epi16(B2, mult);
157
0
        D = _mm256_sub_epi16(A2, C2);
158
0
      }
159
0
      _mm256_storeu_si256((__m256i*)values, D);
160
0
      for (i = 2; i < 32; i += 4) ++histo[values[i]];
161
0
    }
162
0
  }
163
0
  {
164
0
    const int left_over = tile_width & 7;
165
0
    if (left_over > 0) {
166
0
      VP8LCollectColorRedTransforms_SSE(argb + tile_width - left_over, stride,
167
0
                                        left_over, tile_height, green_to_red,
168
0
                                        histo);
169
0
    }
170
0
  }
171
0
}
172
#undef SPAN
173
#undef MK_CST_16
174
175
//------------------------------------------------------------------------------
176
177
// Note we are adding uint32_t's as *signed* int32's (using _mm256_add_epi32).
178
// But that's ok since the histogram values are less than 1<<28 (max picture
179
// size).
180
static void AddVector_AVX2(const uint32_t* WEBP_RESTRICT a,
181
                           const uint32_t* WEBP_RESTRICT b,
182
0
                           uint32_t* WEBP_RESTRICT out, int size) {
183
0
  int i = 0;
184
0
  int aligned_size = size & ~31;
185
  // Size is, at minimum, NUM_DISTANCE_CODES (40) and may be as large as
186
  // NUM_LITERAL_CODES (256) + NUM_LENGTH_CODES (24) + (0 or a non-zero power of
187
  // 2). See the usage in VP8LHistogramAdd().
188
0
  assert(size >= 32);
189
0
  assert(size % 2 == 0);
190
191
0
  do {
192
0
    const __m256i a0 = _mm256_loadu_si256((const __m256i*)&a[i + 0]);
193
0
    const __m256i a1 = _mm256_loadu_si256((const __m256i*)&a[i + 8]);
194
0
    const __m256i a2 = _mm256_loadu_si256((const __m256i*)&a[i + 16]);
195
0
    const __m256i a3 = _mm256_loadu_si256((const __m256i*)&a[i + 24]);
196
0
    const __m256i b0 = _mm256_loadu_si256((const __m256i*)&b[i + 0]);
197
0
    const __m256i b1 = _mm256_loadu_si256((const __m256i*)&b[i + 8]);
198
0
    const __m256i b2 = _mm256_loadu_si256((const __m256i*)&b[i + 16]);
199
0
    const __m256i b3 = _mm256_loadu_si256((const __m256i*)&b[i + 24]);
200
0
    _mm256_storeu_si256((__m256i*)&out[i + 0], _mm256_add_epi32(a0, b0));
201
0
    _mm256_storeu_si256((__m256i*)&out[i + 8], _mm256_add_epi32(a1, b1));
202
0
    _mm256_storeu_si256((__m256i*)&out[i + 16], _mm256_add_epi32(a2, b2));
203
0
    _mm256_storeu_si256((__m256i*)&out[i + 24], _mm256_add_epi32(a3, b3));
204
0
    i += 32;
205
0
  } while (i != aligned_size);
206
207
0
  if ((size & 16) != 0) {
208
0
    const __m256i a0 = _mm256_loadu_si256((const __m256i*)&a[i + 0]);
209
0
    const __m256i a1 = _mm256_loadu_si256((const __m256i*)&a[i + 8]);
210
0
    const __m256i b0 = _mm256_loadu_si256((const __m256i*)&b[i + 0]);
211
0
    const __m256i b1 = _mm256_loadu_si256((const __m256i*)&b[i + 8]);
212
0
    _mm256_storeu_si256((__m256i*)&out[i + 0], _mm256_add_epi32(a0, b0));
213
0
    _mm256_storeu_si256((__m256i*)&out[i + 8], _mm256_add_epi32(a1, b1));
214
0
    i += 16;
215
0
  }
216
217
0
  size &= 15;
218
0
  if (size == 8) {
219
0
    const __m256i a0 = _mm256_loadu_si256((const __m256i*)&a[i]);
220
0
    const __m256i b0 = _mm256_loadu_si256((const __m256i*)&b[i]);
221
0
    _mm256_storeu_si256((__m256i*)&out[i], _mm256_add_epi32(a0, b0));
222
0
  } else {
223
0
    for (; size--; ++i) {
224
0
      out[i] = a[i] + b[i];
225
0
    }
226
0
  }
227
0
}
228
229
static void AddVectorEq_AVX2(const uint32_t* WEBP_RESTRICT a,
230
0
                             uint32_t* WEBP_RESTRICT out, int size) {
231
0
  int i = 0;
232
0
  int aligned_size = size & ~31;
233
  // Size is, at minimum, NUM_DISTANCE_CODES (40) and may be as large as
234
  // NUM_LITERAL_CODES (256) + NUM_LENGTH_CODES (24) + (0 or a non-zero power of
235
  // 2). See the usage in VP8LHistogramAdd().
236
0
  assert(size >= 32);
237
0
  assert(size % 2 == 0);
238
239
0
  do {
240
0
    const __m256i a0 = _mm256_loadu_si256((const __m256i*)&a[i + 0]);
241
0
    const __m256i a1 = _mm256_loadu_si256((const __m256i*)&a[i + 8]);
242
0
    const __m256i a2 = _mm256_loadu_si256((const __m256i*)&a[i + 16]);
243
0
    const __m256i a3 = _mm256_loadu_si256((const __m256i*)&a[i + 24]);
244
0
    const __m256i b0 = _mm256_loadu_si256((const __m256i*)&out[i + 0]);
245
0
    const __m256i b1 = _mm256_loadu_si256((const __m256i*)&out[i + 8]);
246
0
    const __m256i b2 = _mm256_loadu_si256((const __m256i*)&out[i + 16]);
247
0
    const __m256i b3 = _mm256_loadu_si256((const __m256i*)&out[i + 24]);
248
0
    _mm256_storeu_si256((__m256i*)&out[i + 0], _mm256_add_epi32(a0, b0));
249
0
    _mm256_storeu_si256((__m256i*)&out[i + 8], _mm256_add_epi32(a1, b1));
250
0
    _mm256_storeu_si256((__m256i*)&out[i + 16], _mm256_add_epi32(a2, b2));
251
0
    _mm256_storeu_si256((__m256i*)&out[i + 24], _mm256_add_epi32(a3, b3));
252
0
    i += 32;
253
0
  } while (i != aligned_size);
254
255
0
  if ((size & 16) != 0) {
256
0
    const __m256i a0 = _mm256_loadu_si256((const __m256i*)&a[i + 0]);
257
0
    const __m256i a1 = _mm256_loadu_si256((const __m256i*)&a[i + 8]);
258
0
    const __m256i b0 = _mm256_loadu_si256((const __m256i*)&out[i + 0]);
259
0
    const __m256i b1 = _mm256_loadu_si256((const __m256i*)&out[i + 8]);
260
0
    _mm256_storeu_si256((__m256i*)&out[i + 0], _mm256_add_epi32(a0, b0));
261
0
    _mm256_storeu_si256((__m256i*)&out[i + 8], _mm256_add_epi32(a1, b1));
262
0
    i += 16;
263
0
  }
264
265
0
  size &= 15;
266
0
  if (size == 8) {
267
0
    const __m256i a0 = _mm256_loadu_si256((const __m256i*)&a[i]);
268
0
    const __m256i b0 = _mm256_loadu_si256((const __m256i*)&out[i]);
269
0
    _mm256_storeu_si256((__m256i*)&out[i], _mm256_add_epi32(a0, b0));
270
0
  } else {
271
0
    for (; size--; ++i) {
272
0
      out[i] += a[i];
273
0
    }
274
0
  }
275
0
}
276
277
//------------------------------------------------------------------------------
278
// Entropy
279
280
#if !defined(WEBP_HAVE_SLOW_CLZ_CTZ)
281
282
static uint64_t CombinedShannonEntropy_AVX2(const uint32_t X[256],
283
0
                                            const uint32_t Y[256]) {
284
0
  int i;
285
0
  uint64_t retval = 0;
286
0
  uint32_t sumX = 0, sumXY = 0;
287
0
  const __m256i zero = _mm256_setzero_si256();
288
289
0
  for (i = 0; i < 256; i += 32) {
290
0
    const __m256i x0 = _mm256_loadu_si256((const __m256i*)(X + i + 0));
291
0
    const __m256i y0 = _mm256_loadu_si256((const __m256i*)(Y + i + 0));
292
0
    const __m256i x1 = _mm256_loadu_si256((const __m256i*)(X + i + 8));
293
0
    const __m256i y1 = _mm256_loadu_si256((const __m256i*)(Y + i + 8));
294
0
    const __m256i x2 = _mm256_loadu_si256((const __m256i*)(X + i + 16));
295
0
    const __m256i y2 = _mm256_loadu_si256((const __m256i*)(Y + i + 16));
296
0
    const __m256i x3 = _mm256_loadu_si256((const __m256i*)(X + i + 24));
297
0
    const __m256i y3 = _mm256_loadu_si256((const __m256i*)(Y + i + 24));
298
0
    const __m256i x4 = _mm256_packs_epi16(_mm256_packs_epi32(x0, x1),
299
0
                                          _mm256_packs_epi32(x2, x3));
300
0
    const __m256i y4 = _mm256_packs_epi16(_mm256_packs_epi32(y0, y1),
301
0
                                          _mm256_packs_epi32(y2, y3));
302
    // Packed pixels are actually in order: ... 17 16 12 11 10 9 8 3 2 1 0
303
0
    const __m256i x5 = _mm256_permutevar8x32_epi32(
304
0
        x4, _mm256_set_epi32(7, 3, 6, 2, 5, 1, 4, 0));
305
0
    const __m256i y5 = _mm256_permutevar8x32_epi32(
306
0
        y4, _mm256_set_epi32(7, 3, 6, 2, 5, 1, 4, 0));
307
0
    const uint32_t mx =
308
0
        (uint32_t)_mm256_movemask_epi8(_mm256_cmpgt_epi8(x5, zero));
309
0
    uint32_t my =
310
0
        (uint32_t)_mm256_movemask_epi8(_mm256_cmpgt_epi8(y5, zero)) | mx;
311
0
    while (my) {
312
0
      const int32_t j = BitsCtz(my);
313
0
      uint32_t xy;
314
0
      if ((mx >> j) & 1) {
315
0
        const int x = X[i + j];
316
0
        sumXY += x;
317
0
        retval += VP8LFastSLog2(x);
318
0
      }
319
0
      xy = X[i + j] + Y[i + j];
320
0
      sumX += xy;
321
0
      retval += VP8LFastSLog2(xy);
322
0
      my &= my - 1;
323
0
    }
324
0
  }
325
0
  retval = VP8LFastSLog2(sumX) + VP8LFastSLog2(sumXY) - retval;
326
0
  return retval;
327
0
}
328
329
#else
330
331
#define DONT_USE_COMBINED_SHANNON_ENTROPY_SSE2_FUNC  // won't be faster
332
333
#endif
334
335
//------------------------------------------------------------------------------
336
337
static int VectorMismatch_AVX2(const uint32_t* const array1,
338
0
                               const uint32_t* const array2, int length) {
339
0
  int match_len;
340
341
0
  if (length >= 24) {
342
0
    __m256i A0 = _mm256_loadu_si256((const __m256i*)&array1[0]);
343
0
    __m256i A1 = _mm256_loadu_si256((const __m256i*)&array2[0]);
344
0
    match_len = 0;
345
0
    do {
346
      // Loop unrolling and early load both provide a speedup of 10% for the
347
      // current function. Also, max_limit can be MAX_LENGTH=4096 at most.
348
0
      const __m256i cmpA = _mm256_cmpeq_epi32(A0, A1);
349
0
      const __m256i B0 =
350
0
          _mm256_loadu_si256((const __m256i*)&array1[match_len + 8]);
351
0
      const __m256i B1 =
352
0
          _mm256_loadu_si256((const __m256i*)&array2[match_len + 8]);
353
0
      if ((uint32_t)_mm256_movemask_epi8(cmpA) != 0xffffffff) break;
354
0
      match_len += 8;
355
356
0
      {
357
0
        const __m256i cmpB = _mm256_cmpeq_epi32(B0, B1);
358
0
        A0 = _mm256_loadu_si256((const __m256i*)&array1[match_len + 8]);
359
0
        A1 = _mm256_loadu_si256((const __m256i*)&array2[match_len + 8]);
360
0
        if ((uint32_t)_mm256_movemask_epi8(cmpB) != 0xffffffff) break;
361
0
        match_len += 8;
362
0
      }
363
0
    } while (match_len + 24 < length);
364
0
  } else {
365
0
    match_len = 0;
366
    // Unroll the potential first two loops.
367
0
    if (length >= 8 &&
368
0
        (uint32_t)_mm256_movemask_epi8(_mm256_cmpeq_epi32(
369
0
            _mm256_loadu_si256((const __m256i*)&array1[0]),
370
0
            _mm256_loadu_si256((const __m256i*)&array2[0]))) == 0xffffffff) {
371
0
      match_len = 8;
372
0
      if (length >= 16 &&
373
0
          (uint32_t)_mm256_movemask_epi8(_mm256_cmpeq_epi32(
374
0
              _mm256_loadu_si256((const __m256i*)&array1[8]),
375
0
              _mm256_loadu_si256((const __m256i*)&array2[8]))) == 0xffffffff) {
376
0
        match_len = 16;
377
0
      }
378
0
    }
379
0
  }
380
381
0
  while (match_len < length && array1[match_len] == array2[match_len]) {
382
0
    ++match_len;
383
0
  }
384
0
  return match_len;
385
0
}
386
387
// Bundles multiple (1, 2, 4 or 8) pixels into a single pixel.
388
static void BundleColorMap_AVX2(const uint8_t* WEBP_RESTRICT const row,
389
                                int width, int xbits,
390
0
                                uint32_t* WEBP_RESTRICT dst) {
391
0
  int x = 0;
392
0
  assert(xbits >= 0);
393
0
  assert(xbits <= 3);
394
0
  switch (xbits) {
395
0
    case 0: {
396
0
      const __m256i ff = _mm256_set1_epi16((short)0xff00);
397
0
      const __m256i zero = _mm256_setzero_si256();
398
      // Store 0xff000000 | (row[x] << 8).
399
0
      for (x = 0; x + 32 <= width; x += 32, dst += 32) {
400
0
        const __m256i in = _mm256_loadu_si256((const __m256i*)&row[x]);
401
0
        const __m256i in_lo = _mm256_unpacklo_epi8(zero, in);
402
0
        const __m256i dst0 = _mm256_unpacklo_epi16(in_lo, ff);
403
0
        const __m256i dst1 = _mm256_unpackhi_epi16(in_lo, ff);
404
0
        const __m256i in_hi = _mm256_unpackhi_epi8(zero, in);
405
0
        const __m256i dst2 = _mm256_unpacklo_epi16(in_hi, ff);
406
0
        const __m256i dst3 = _mm256_unpackhi_epi16(in_hi, ff);
407
0
        _mm256_storeu2_m128i((__m128i*)&dst[16], (__m128i*)&dst[0], dst0);
408
0
        _mm256_storeu2_m128i((__m128i*)&dst[20], (__m128i*)&dst[4], dst1);
409
0
        _mm256_storeu2_m128i((__m128i*)&dst[24], (__m128i*)&dst[8], dst2);
410
0
        _mm256_storeu2_m128i((__m128i*)&dst[28], (__m128i*)&dst[12], dst3);
411
0
      }
412
0
      break;
413
0
    }
414
0
    case 1: {
415
0
      const __m256i ff = _mm256_set1_epi16((short)0xff00);
416
0
      const __m256i mul = _mm256_set1_epi16(0x110);
417
0
      for (x = 0; x + 32 <= width; x += 32, dst += 16) {
418
        // 0a0b | (where a/b are 4 bits).
419
0
        const __m256i in = _mm256_loadu_si256((const __m256i*)&row[x]);
420
0
        const __m256i tmp = _mm256_mullo_epi16(in, mul);  // aba0
421
0
        const __m256i pack = _mm256_and_si256(tmp, ff);   // ab00
422
0
        const __m256i dst0 = _mm256_unpacklo_epi16(pack, ff);
423
0
        const __m256i dst1 = _mm256_unpackhi_epi16(pack, ff);
424
0
        _mm256_storeu2_m128i((__m128i*)&dst[8], (__m128i*)&dst[0], dst0);
425
0
        _mm256_storeu2_m128i((__m128i*)&dst[12], (__m128i*)&dst[4], dst1);
426
0
      }
427
0
      break;
428
0
    }
429
0
    case 2: {
430
0
      const __m256i mask_or = _mm256_set1_epi32((int)0xff000000);
431
0
      const __m256i mul_cst = _mm256_set1_epi16(0x0104);
432
0
      const __m256i mask_mul = _mm256_set1_epi16(0x0f00);
433
0
      for (x = 0; x + 32 <= width; x += 32, dst += 8) {
434
        // 000a000b000c000d | (where a/b/c/d are 2 bits).
435
0
        const __m256i in = _mm256_loadu_si256((const __m256i*)&row[x]);
436
0
        const __m256i mul =
437
0
            _mm256_mullo_epi16(in, mul_cst);  // 00ab00b000cd00d0
438
0
        const __m256i tmp =
439
0
            _mm256_and_si256(mul, mask_mul);               //  00ab000000cd0000
440
0
        const __m256i shift = _mm256_srli_epi32(tmp, 12);  // 00000000ab000000
441
0
        const __m256i pack = _mm256_or_si256(shift, tmp);  // 00000000abcd0000
442
        // Convert to 0xff00**00.
443
0
        const __m256i res = _mm256_or_si256(pack, mask_or);
444
0
        _mm256_storeu_si256((__m256i*)dst, res);
445
0
      }
446
0
      break;
447
0
    }
448
0
    default: {
449
0
      assert(xbits == 3);
450
0
      for (x = 0; x + 32 <= width; x += 32, dst += 4) {
451
        // 0000000a00000000b... | (where a/b are 1 bit).
452
0
        const __m256i in = _mm256_loadu_si256((const __m256i*)&row[x]);
453
0
        const __m256i shift = _mm256_slli_epi64(in, 7);
454
0
        const uint32_t move = _mm256_movemask_epi8(shift);
455
0
        dst[0] = 0xff000000 | ((move & 0xff) << 8);
456
0
        dst[1] = 0xff000000 | (move & 0xff00);
457
0
        dst[2] = 0xff000000 | ((move & 0xff0000) >> 8);
458
0
        dst[3] = 0xff000000 | ((move & 0xff000000) >> 16);
459
0
      }
460
0
      break;
461
0
    }
462
0
  }
463
0
  if (x != width) {
464
0
    VP8LBundleColorMap_SSE(row + x, width - x, xbits, dst);
465
0
  }
466
0
}
467
468
//------------------------------------------------------------------------------
469
// Batch version of Predictor Transform subtraction
470
471
static WEBP_INLINE void Average2_m256i(const __m256i* const a0,
472
                                       const __m256i* const a1,
473
0
                                       __m256i* const avg) {
474
  // (a + b) >> 1 = ((a + b + 1) >> 1) - ((a ^ b) & 1)
475
0
  const __m256i ones = _mm256_set1_epi8(1);
476
0
  const __m256i avg1 = _mm256_avg_epu8(*a0, *a1);
477
0
  const __m256i one = _mm256_and_si256(_mm256_xor_si256(*a0, *a1), ones);
478
0
  *avg = _mm256_sub_epi8(avg1, one);
479
0
}
480
481
// Predictor0: ARGB_BLACK.
482
static void PredictorSub0_AVX2(const uint32_t* in, const uint32_t* upper,
483
0
                               int num_pixels, uint32_t* WEBP_RESTRICT out) {
484
0
  int i;
485
0
  const __m256i black = _mm256_set1_epi32((int)ARGB_BLACK);
486
0
  for (i = 0; i + 8 <= num_pixels; i += 8) {
487
0
    const __m256i src = _mm256_loadu_si256((const __m256i*)&in[i]);
488
0
    const __m256i res = _mm256_sub_epi8(src, black);
489
0
    _mm256_storeu_si256((__m256i*)&out[i], res);
490
0
  }
491
0
  if (i != num_pixels) {
492
0
    VP8LPredictorsSub_SSE[0](in + i, NULL, num_pixels - i, out + i);
493
0
  }
494
0
  (void)upper;
495
0
}
496
497
#define GENERATE_PREDICTOR_1(X, IN)                                          \
498
  static void PredictorSub##X##_AVX2(                                        \
499
      const uint32_t* const in, const uint32_t* const upper, int num_pixels, \
500
0
      uint32_t* WEBP_RESTRICT const out) {                                   \
501
0
    int i;                                                                   \
502
0
    for (i = 0; i + 8 <= num_pixels; i += 8) {                               \
503
0
      const __m256i src = _mm256_loadu_si256((const __m256i*)&in[i]);        \
504
0
      const __m256i pred = _mm256_loadu_si256((const __m256i*)&(IN));        \
505
0
      const __m256i res = _mm256_sub_epi8(src, pred);                        \
506
0
      _mm256_storeu_si256((__m256i*)&out[i], res);                           \
507
0
    }                                                                        \
508
0
    if (i != num_pixels) {                                                   \
509
0
      VP8LPredictorsSub_SSE[(X)](in + i, WEBP_OFFSET_PTR(upper, i),          \
510
0
                                 num_pixels - i, out + i);                   \
511
0
    }                                                                        \
512
0
  }
513
514
0
GENERATE_PREDICTOR_1(1, in[i - 1])     // Predictor1: L
515
0
GENERATE_PREDICTOR_1(2, upper[i])      // Predictor2: T
516
0
GENERATE_PREDICTOR_1(3, upper[i + 1])  // Predictor3: TR
517
0
GENERATE_PREDICTOR_1(4, upper[i - 1])  // Predictor4: TL
518
#undef GENERATE_PREDICTOR_1
519
520
// Predictor5: avg2(avg2(L, TR), T)
521
static void PredictorSub5_AVX2(const uint32_t* in, const uint32_t* upper,
522
0
                               int num_pixels, uint32_t* WEBP_RESTRICT out) {
523
0
  int i;
524
0
  for (i = 0; i + 8 <= num_pixels; i += 8) {
525
0
    const __m256i L = _mm256_loadu_si256((const __m256i*)&in[i - 1]);
526
0
    const __m256i T = _mm256_loadu_si256((const __m256i*)&upper[i]);
527
0
    const __m256i TR = _mm256_loadu_si256((const __m256i*)&upper[i + 1]);
528
0
    const __m256i src = _mm256_loadu_si256((const __m256i*)&in[i]);
529
0
    __m256i avg, pred, res;
530
0
    Average2_m256i(&L, &TR, &avg);
531
0
    Average2_m256i(&avg, &T, &pred);
532
0
    res = _mm256_sub_epi8(src, pred);
533
0
    _mm256_storeu_si256((__m256i*)&out[i], res);
534
0
  }
535
0
  if (i != num_pixels) {
536
0
    VP8LPredictorsSub_SSE[5](in + i, upper + i, num_pixels - i, out + i);
537
0
  }
538
0
}
539
540
#define GENERATE_PREDICTOR_2(X, A, B)                                         \
541
  static void PredictorSub##X##_AVX2(const uint32_t* in,                      \
542
                                     const uint32_t* upper, int num_pixels,   \
543
0
                                     uint32_t* WEBP_RESTRICT out) {           \
544
0
    int i;                                                                    \
545
0
    for (i = 0; i + 8 <= num_pixels; i += 8) {                                \
546
0
      const __m256i tA = _mm256_loadu_si256((const __m256i*)&(A));            \
547
0
      const __m256i tB = _mm256_loadu_si256((const __m256i*)&(B));            \
548
0
      const __m256i src = _mm256_loadu_si256((const __m256i*)&in[i]);         \
549
0
      __m256i pred, res;                                                      \
550
0
      Average2_m256i(&tA, &tB, &pred);                                        \
551
0
      res = _mm256_sub_epi8(src, pred);                                       \
552
0
      _mm256_storeu_si256((__m256i*)&out[i], res);                            \
553
0
    }                                                                         \
554
0
    if (i != num_pixels) {                                                    \
555
0
      VP8LPredictorsSub_SSE[(X)](in + i, upper + i, num_pixels - i, out + i); \
556
0
    }                                                                         \
557
0
  }
Unexecuted instantiation: lossless_enc_avx2.c:PredictorSub6_AVX2
Unexecuted instantiation: lossless_enc_avx2.c:PredictorSub7_AVX2
Unexecuted instantiation: lossless_enc_avx2.c:PredictorSub8_AVX2
Unexecuted instantiation: lossless_enc_avx2.c:PredictorSub9_AVX2
558
559
GENERATE_PREDICTOR_2(6, in[i - 1], upper[i - 1])  // Predictor6: avg(L, TL)
560
GENERATE_PREDICTOR_2(7, in[i - 1], upper[i])      // Predictor7: avg(L, T)
561
GENERATE_PREDICTOR_2(8, upper[i - 1], upper[i])   // Predictor8: avg(TL, T)
562
GENERATE_PREDICTOR_2(9, upper[i], upper[i + 1])   // Predictor9: average(T, TR)
563
#undef GENERATE_PREDICTOR_2
564
565
// Predictor10: avg(avg(L,TL), avg(T, TR)).
566
static void PredictorSub10_AVX2(const uint32_t* in, const uint32_t* upper,
567
0
                                int num_pixels, uint32_t* WEBP_RESTRICT out) {
568
0
  int i;
569
0
  for (i = 0; i + 8 <= num_pixels; i += 8) {
570
0
    const __m256i L = _mm256_loadu_si256((const __m256i*)&in[i - 1]);
571
0
    const __m256i src = _mm256_loadu_si256((const __m256i*)&in[i]);
572
0
    const __m256i TL = _mm256_loadu_si256((const __m256i*)&upper[i - 1]);
573
0
    const __m256i T = _mm256_loadu_si256((const __m256i*)&upper[i]);
574
0
    const __m256i TR = _mm256_loadu_si256((const __m256i*)&upper[i + 1]);
575
0
    __m256i avgTTR, avgLTL, avg, res;
576
0
    Average2_m256i(&T, &TR, &avgTTR);
577
0
    Average2_m256i(&L, &TL, &avgLTL);
578
0
    Average2_m256i(&avgTTR, &avgLTL, &avg);
579
0
    res = _mm256_sub_epi8(src, avg);
580
0
    _mm256_storeu_si256((__m256i*)&out[i], res);
581
0
  }
582
0
  if (i != num_pixels) {
583
0
    VP8LPredictorsSub_SSE[10](in + i, upper + i, num_pixels - i, out + i);
584
0
  }
585
0
}
586
587
// Predictor11: select.
588
static void GetSumAbsDiff32_AVX2(const __m256i* const A, const __m256i* const B,
589
0
                                 __m256i* const out) {
590
  // We can unpack with any value on the upper 32 bits, provided it's the same
591
  // on both operands (to that their sum of abs diff is zero). Here we use *A.
592
0
  const __m256i A_lo = _mm256_unpacklo_epi32(*A, *A);
593
0
  const __m256i B_lo = _mm256_unpacklo_epi32(*B, *A);
594
0
  const __m256i A_hi = _mm256_unpackhi_epi32(*A, *A);
595
0
  const __m256i B_hi = _mm256_unpackhi_epi32(*B, *A);
596
0
  const __m256i s_lo = _mm256_sad_epu8(A_lo, B_lo);
597
0
  const __m256i s_hi = _mm256_sad_epu8(A_hi, B_hi);
598
0
  *out = _mm256_packs_epi32(s_lo, s_hi);
599
0
}
600
601
static void PredictorSub11_AVX2(const uint32_t* in, const uint32_t* upper,
602
0
                                int num_pixels, uint32_t* WEBP_RESTRICT out) {
603
0
  int i;
604
0
  for (i = 0; i + 8 <= num_pixels; i += 8) {
605
0
    const __m256i L = _mm256_loadu_si256((const __m256i*)&in[i - 1]);
606
0
    const __m256i T = _mm256_loadu_si256((const __m256i*)&upper[i]);
607
0
    const __m256i TL = _mm256_loadu_si256((const __m256i*)&upper[i - 1]);
608
0
    const __m256i src = _mm256_loadu_si256((const __m256i*)&in[i]);
609
0
    __m256i pa, pb;
610
0
    GetSumAbsDiff32_AVX2(&T, &TL, &pa);  // pa = sum |T-TL|
611
0
    GetSumAbsDiff32_AVX2(&L, &TL, &pb);  // pb = sum |L-TL|
612
0
    {
613
0
      const __m256i mask = _mm256_cmpgt_epi32(pb, pa);
614
0
      const __m256i A = _mm256_and_si256(mask, L);
615
0
      const __m256i B = _mm256_andnot_si256(mask, T);
616
0
      const __m256i pred = _mm256_or_si256(A, B);  // pred = (L > T)? L : T
617
0
      const __m256i res = _mm256_sub_epi8(src, pred);
618
0
      _mm256_storeu_si256((__m256i*)&out[i], res);
619
0
    }
620
0
  }
621
0
  if (i != num_pixels) {
622
0
    VP8LPredictorsSub_SSE[11](in + i, upper + i, num_pixels - i, out + i);
623
0
  }
624
0
}
625
626
// Predictor12: ClampedSubSubtractFull.
627
static void PredictorSub12_AVX2(const uint32_t* in, const uint32_t* upper,
628
0
                                int num_pixels, uint32_t* WEBP_RESTRICT out) {
629
0
  int i;
630
0
  const __m256i zero = _mm256_setzero_si256();
631
0
  for (i = 0; i + 8 <= num_pixels; i += 8) {
632
0
    const __m256i src = _mm256_loadu_si256((const __m256i*)&in[i]);
633
0
    const __m256i L = _mm256_loadu_si256((const __m256i*)&in[i - 1]);
634
0
    const __m256i L_lo = _mm256_unpacklo_epi8(L, zero);
635
0
    const __m256i L_hi = _mm256_unpackhi_epi8(L, zero);
636
0
    const __m256i T = _mm256_loadu_si256((const __m256i*)&upper[i]);
637
0
    const __m256i T_lo = _mm256_unpacklo_epi8(T, zero);
638
0
    const __m256i T_hi = _mm256_unpackhi_epi8(T, zero);
639
0
    const __m256i TL = _mm256_loadu_si256((const __m256i*)&upper[i - 1]);
640
0
    const __m256i TL_lo = _mm256_unpacklo_epi8(TL, zero);
641
0
    const __m256i TL_hi = _mm256_unpackhi_epi8(TL, zero);
642
0
    const __m256i diff_lo = _mm256_sub_epi16(T_lo, TL_lo);
643
0
    const __m256i diff_hi = _mm256_sub_epi16(T_hi, TL_hi);
644
0
    const __m256i pred_lo = _mm256_add_epi16(L_lo, diff_lo);
645
0
    const __m256i pred_hi = _mm256_add_epi16(L_hi, diff_hi);
646
0
    const __m256i pred = _mm256_packus_epi16(pred_lo, pred_hi);
647
0
    const __m256i res = _mm256_sub_epi8(src, pred);
648
0
    _mm256_storeu_si256((__m256i*)&out[i], res);
649
0
  }
650
0
  if (i != num_pixels) {
651
0
    VP8LPredictorsSub_SSE[12](in + i, upper + i, num_pixels - i, out + i);
652
0
  }
653
0
}
654
655
// Predictors13: ClampedAddSubtractHalf
656
static void PredictorSub13_AVX2(const uint32_t* in, const uint32_t* upper,
657
0
                                int num_pixels, uint32_t* WEBP_RESTRICT out) {
658
0
  int i;
659
0
  const __m256i zero = _mm256_setzero_si256();
660
0
  for (i = 0; i + 8 <= num_pixels; i += 8) {
661
0
    const __m256i L = _mm256_loadu_si256((const __m256i*)&in[i - 1]);
662
0
    const __m256i src = _mm256_loadu_si256((const __m256i*)&in[i]);
663
0
    const __m256i T = _mm256_loadu_si256((const __m256i*)&upper[i]);
664
0
    const __m256i TL = _mm256_loadu_si256((const __m256i*)&upper[i - 1]);
665
    // lo.
666
0
    const __m256i L_lo = _mm256_unpacklo_epi8(L, zero);
667
0
    const __m256i T_lo = _mm256_unpacklo_epi8(T, zero);
668
0
    const __m256i TL_lo = _mm256_unpacklo_epi8(TL, zero);
669
0
    const __m256i sum_lo = _mm256_add_epi16(T_lo, L_lo);
670
0
    const __m256i avg_lo = _mm256_srli_epi16(sum_lo, 1);
671
0
    const __m256i A1_lo = _mm256_sub_epi16(avg_lo, TL_lo);
672
0
    const __m256i bit_fix_lo = _mm256_cmpgt_epi16(TL_lo, avg_lo);
673
0
    const __m256i A2_lo = _mm256_sub_epi16(A1_lo, bit_fix_lo);
674
0
    const __m256i A3_lo = _mm256_srai_epi16(A2_lo, 1);
675
0
    const __m256i A4_lo = _mm256_add_epi16(avg_lo, A3_lo);
676
    // hi.
677
0
    const __m256i L_hi = _mm256_unpackhi_epi8(L, zero);
678
0
    const __m256i T_hi = _mm256_unpackhi_epi8(T, zero);
679
0
    const __m256i TL_hi = _mm256_unpackhi_epi8(TL, zero);
680
0
    const __m256i sum_hi = _mm256_add_epi16(T_hi, L_hi);
681
0
    const __m256i avg_hi = _mm256_srli_epi16(sum_hi, 1);
682
0
    const __m256i A1_hi = _mm256_sub_epi16(avg_hi, TL_hi);
683
0
    const __m256i bit_fix_hi = _mm256_cmpgt_epi16(TL_hi, avg_hi);
684
0
    const __m256i A2_hi = _mm256_sub_epi16(A1_hi, bit_fix_hi);
685
0
    const __m256i A3_hi = _mm256_srai_epi16(A2_hi, 1);
686
0
    const __m256i A4_hi = _mm256_add_epi16(avg_hi, A3_hi);
687
688
0
    const __m256i pred = _mm256_packus_epi16(A4_lo, A4_hi);
689
0
    const __m256i res = _mm256_sub_epi8(src, pred);
690
0
    _mm256_storeu_si256((__m256i*)&out[i], res);
691
0
  }
692
0
  if (i != num_pixels) {
693
0
    VP8LPredictorsSub_SSE[13](in + i, upper + i, num_pixels - i, out + i);
694
0
  }
695
0
}
696
697
//------------------------------------------------------------------------------
698
// Entry point
699
700
extern void VP8LEncDspInitAVX2(void);
701
702
0
WEBP_TSAN_IGNORE_FUNCTION void VP8LEncDspInitAVX2(void) {
703
0
  VP8LSubtractGreenFromBlueAndRed = SubtractGreenFromBlueAndRed_AVX2;
704
0
  VP8LTransformColor = TransformColor_AVX2;
705
0
  VP8LCollectColorBlueTransforms = CollectColorBlueTransforms_AVX2;
706
0
  VP8LCollectColorRedTransforms = CollectColorRedTransforms_AVX2;
707
0
  VP8LAddVector = AddVector_AVX2;
708
0
  VP8LAddVectorEq = AddVectorEq_AVX2;
709
0
  VP8LCombinedShannonEntropy = CombinedShannonEntropy_AVX2;
710
0
  VP8LVectorMismatch = VectorMismatch_AVX2;
711
0
  VP8LBundleColorMap = BundleColorMap_AVX2;
712
713
0
  VP8LPredictorsSub[0] = PredictorSub0_AVX2;
714
0
  VP8LPredictorsSub[1] = PredictorSub1_AVX2;
715
0
  VP8LPredictorsSub[2] = PredictorSub2_AVX2;
716
0
  VP8LPredictorsSub[3] = PredictorSub3_AVX2;
717
0
  VP8LPredictorsSub[4] = PredictorSub4_AVX2;
718
0
  VP8LPredictorsSub[5] = PredictorSub5_AVX2;
719
0
  VP8LPredictorsSub[6] = PredictorSub6_AVX2;
720
0
  VP8LPredictorsSub[7] = PredictorSub7_AVX2;
721
0
  VP8LPredictorsSub[8] = PredictorSub8_AVX2;
722
0
  VP8LPredictorsSub[9] = PredictorSub9_AVX2;
723
0
  VP8LPredictorsSub[10] = PredictorSub10_AVX2;
724
0
  VP8LPredictorsSub[11] = PredictorSub11_AVX2;
725
0
  VP8LPredictorsSub[12] = PredictorSub12_AVX2;
726
0
  VP8LPredictorsSub[13] = PredictorSub13_AVX2;
727
0
  VP8LPredictorsSub[14] = PredictorSub0_AVX2;  // <- padding security sentinels
728
0
  VP8LPredictorsSub[15] = PredictorSub0_AVX2;
729
0
}
730
731
#else  // !WEBP_USE_AVX2
732
733
WEBP_DSP_INIT_STUB(VP8LEncDspInitAVX2)
734
735
#endif  // WEBP_USE_AVX2