Coverage Report

Created: 2026-09-28 10:59

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/work/workdir/UnpackedTarball/highway/hwy/targets.cc
Line
Count
Source
1
// Copyright 2019 Google LLC
2
// Copyright 2025 Arm Limited and/or its affiliates <open-source-office@arm.com>
3
// SPDX-License-Identifier: Apache-2.0
4
//
5
// Licensed under the Apache License, Version 2.0 (the "License");
6
// you may not use this file except in compliance with the License.
7
// You may obtain a copy of the License at
8
//
9
//      http://www.apache.org/licenses/LICENSE-2.0
10
//
11
// Unless required by applicable law or agreed to in writing, software
12
// distributed under the License is distributed on an "AS IS" BASIS,
13
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
14
// See the License for the specific language governing permissions and
15
// limitations under the License.
16
17
#include "hwy/targets.h"
18
19
#include <stdint.h>
20
#include <stdio.h>
21
22
#include "hwy/base.h"
23
#include "hwy/detect_targets.h"
24
#include "hwy/highway.h"
25
#include "hwy/x86_cpuid.h"
26
27
#if HWY_ARCH_X86
28
#include <xmmintrin.h>
29
30
#elif (HWY_ARCH_ARM || HWY_ARCH_PPC || HWY_ARCH_S390X || HWY_ARCH_RISCV || \
31
       HWY_ARCH_LOONGARCH) &&                                              \
32
    HWY_OS_LINUX
33
// sys/auxv.h does not always include asm/hwcap.h, or define HWCAP*, hence we
34
// still include this directly. See #1199.
35
#if HWY_HAVE_ASM_HWCAP
36
#include <asm/hwcap.h>
37
#endif
38
#if HWY_HAVE_AUXV
39
#include <sys/auxv.h>
40
#endif
41
42
#endif  // HWY_ARCH_*
43
44
#if HWY_OS_APPLE
45
#include <sys/sysctl.h>
46
#include <sys/utsname.h>
47
#endif  // HWY_OS_APPLE
48
49
namespace hwy {
50
51
#if HWY_OS_APPLE
52
static HWY_INLINE HWY_MAYBE_UNUSED bool HasCpuFeature(
53
    const char* feature_name) {
54
  int result = 0;
55
  size_t len = sizeof(int);
56
  return (sysctlbyname(feature_name, &result, &len, nullptr, 0) == 0 &&
57
          result != 0);
58
}
59
60
static HWY_INLINE HWY_MAYBE_UNUSED bool ParseU32(const char*& ptr,
61
                                                 uint32_t& parsed_val) {
62
  uint64_t parsed_u64 = 0;
63
64
  const char* start_ptr = ptr;
65
  for (char ch; (ch = (*ptr)) != '\0'; ++ptr) {
66
    unsigned digit = static_cast<unsigned>(static_cast<unsigned char>(ch)) -
67
                     static_cast<unsigned>(static_cast<unsigned char>('0'));
68
    if (digit > 9u) {
69
      break;
70
    }
71
72
    parsed_u64 = (parsed_u64 * 10u) + digit;
73
    if (parsed_u64 > 0xFFFFFFFFu) {
74
      return false;
75
    }
76
  }
77
78
  parsed_val = static_cast<uint32_t>(parsed_u64);
79
  return (ptr != start_ptr);
80
}
81
82
static HWY_INLINE HWY_MAYBE_UNUSED bool IsMacOs12_2OrLater() {
83
  utsname uname_buf;
84
  ZeroBytes(&uname_buf, sizeof(utsname));
85
86
  if ((uname(&uname_buf)) != 0) {
87
    return false;
88
  }
89
90
  const char* ptr = uname_buf.release;
91
  if (!ptr) {
92
    return false;
93
  }
94
95
  uint32_t major;
96
  uint32_t minor;
97
  if (!ParseU32(ptr, major)) {
98
    return false;
99
  }
100
101
  if (*ptr != '.') {
102
    return false;
103
  }
104
105
  ++ptr;
106
  if (!ParseU32(ptr, minor)) {
107
    return false;
108
  }
109
110
  // We are running on macOS 12.2 or later if the Darwin kernel version is 21.3
111
  // or later
112
  return (major > 21 || (major == 21 && minor >= 3));
113
}
114
#endif  // HWY_OS_APPLE
115
116
#if HWY_ARCH_X86 && HWY_HAVE_RUNTIME_DISPATCH
117
namespace x86 {
118
119
// Returns the lower 32 bits of extended control register 0.
120
// Requires CPU support for "OSXSAVE" (see below).
121
0
static uint32_t ReadXCR0() {
122
#if HWY_COMPILER_MSVC
123
  return static_cast<uint32_t>(_xgetbv(0));
124
#else   // HWY_COMPILER_MSVC
125
0
  uint32_t xcr0, xcr0_high;
126
0
  const uint32_t index = 0;
127
0
  asm volatile(".byte 0x0F, 0x01, 0xD0"
128
0
               : "=a"(xcr0), "=d"(xcr0_high)
129
0
               : "c"(index));
130
0
  return xcr0;
131
0
#endif  // HWY_COMPILER_MSVC
132
0
}
133
134
// Arbitrary bit indices indicating which instruction set extensions are
135
// supported. Use enum to ensure values are distinct.
136
enum class FeatureIndex : uint32_t {
137
  kSSE = 0,
138
  kSSE2,
139
  kSSE3,
140
  kSSSE3,
141
142
  kSSE41,
143
  kSSE42,
144
  kCLMUL,
145
  kAES,
146
147
  kAVX,
148
  kAVX2,
149
  kF16C,
150
  kFMA,
151
  kLZCNT,
152
  kBMI,
153
  kBMI2,
154
155
  kAVX512F,
156
  kAVX512VL,
157
  kAVX512CD,
158
  kAVX512DQ,
159
  kAVX512BW,
160
  kAVX512FP16,
161
  kAVX512BF16,
162
163
  kVNNI,
164
  kVPCLMULQDQ,
165
  kVBMI,
166
  kVBMI2,
167
  kVAES,
168
  kPOPCNTDQ,
169
  kBITALG,
170
  kGFNI,
171
172
  kAVX10,
173
  kAPX,
174
175
  kSentinel
176
};
177
static_assert(static_cast<size_t>(FeatureIndex::kSentinel) < 64,
178
              "Too many bits for u64");
179
180
0
static HWY_INLINE constexpr uint64_t Bit(FeatureIndex index) {
181
0
  return 1ull << static_cast<size_t>(index);
182
0
}
183
184
// Returns bit array of FeatureIndex from CPUID feature flags.
185
0
static uint64_t FlagsFromCPUID() {
186
0
  uint64_t flags = 0;  // return value
187
0
  uint32_t abcd[4];
188
0
  Cpuid(0, 0, abcd);
189
0
  const uint32_t max_level = abcd[0];
190
191
  // Standard feature flags
192
0
  Cpuid(1, 0, abcd);
193
0
  flags |= IsBitSet(abcd[3], 25) ? Bit(FeatureIndex::kSSE) : 0;
194
0
  flags |= IsBitSet(abcd[3], 26) ? Bit(FeatureIndex::kSSE2) : 0;
195
0
  flags |= IsBitSet(abcd[2], 0) ? Bit(FeatureIndex::kSSE3) : 0;
196
0
  flags |= IsBitSet(abcd[2], 1) ? Bit(FeatureIndex::kCLMUL) : 0;
197
0
  flags |= IsBitSet(abcd[2], 9) ? Bit(FeatureIndex::kSSSE3) : 0;
198
0
  flags |= IsBitSet(abcd[2], 12) ? Bit(FeatureIndex::kFMA) : 0;
199
0
  flags |= IsBitSet(abcd[2], 19) ? Bit(FeatureIndex::kSSE41) : 0;
200
0
  flags |= IsBitSet(abcd[2], 20) ? Bit(FeatureIndex::kSSE42) : 0;
201
0
  flags |= IsBitSet(abcd[2], 25) ? Bit(FeatureIndex::kAES) : 0;
202
0
  flags |= IsBitSet(abcd[2], 28) ? Bit(FeatureIndex::kAVX) : 0;
203
0
  flags |= IsBitSet(abcd[2], 29) ? Bit(FeatureIndex::kF16C) : 0;
204
205
  // Extended feature flags
206
0
  Cpuid(0x80000001U, 0, abcd);
207
0
  flags |= IsBitSet(abcd[2], 5) ? Bit(FeatureIndex::kLZCNT) : 0;
208
209
  // Extended features
210
0
  if (max_level >= 7) {
211
0
    Cpuid(7, 0, abcd);
212
0
    flags |= IsBitSet(abcd[1], 3) ? Bit(FeatureIndex::kBMI) : 0;
213
0
    flags |= IsBitSet(abcd[1], 5) ? Bit(FeatureIndex::kAVX2) : 0;
214
0
    flags |= IsBitSet(abcd[1], 8) ? Bit(FeatureIndex::kBMI2) : 0;
215
216
0
    flags |= IsBitSet(abcd[1], 16) ? Bit(FeatureIndex::kAVX512F) : 0;
217
0
    flags |= IsBitSet(abcd[1], 17) ? Bit(FeatureIndex::kAVX512DQ) : 0;
218
0
    flags |= IsBitSet(abcd[1], 28) ? Bit(FeatureIndex::kAVX512CD) : 0;
219
0
    flags |= IsBitSet(abcd[1], 30) ? Bit(FeatureIndex::kAVX512BW) : 0;
220
0
    flags |= IsBitSet(abcd[1], 31) ? Bit(FeatureIndex::kAVX512VL) : 0;
221
222
0
    flags |= IsBitSet(abcd[2], 1) ? Bit(FeatureIndex::kVBMI) : 0;
223
0
    flags |= IsBitSet(abcd[2], 6) ? Bit(FeatureIndex::kVBMI2) : 0;
224
0
    flags |= IsBitSet(abcd[2], 8) ? Bit(FeatureIndex::kGFNI) : 0;
225
0
    flags |= IsBitSet(abcd[2], 9) ? Bit(FeatureIndex::kVAES) : 0;
226
0
    flags |= IsBitSet(abcd[2], 10) ? Bit(FeatureIndex::kVPCLMULQDQ) : 0;
227
0
    flags |= IsBitSet(abcd[2], 11) ? Bit(FeatureIndex::kVNNI) : 0;
228
0
    flags |= IsBitSet(abcd[2], 12) ? Bit(FeatureIndex::kBITALG) : 0;
229
0
    flags |= IsBitSet(abcd[2], 14) ? Bit(FeatureIndex::kPOPCNTDQ) : 0;
230
231
0
    flags |= IsBitSet(abcd[3], 23) ? Bit(FeatureIndex::kAVX512FP16) : 0;
232
233
0
    Cpuid(7, 1, abcd);
234
0
    flags |= IsBitSet(abcd[0], 5) ? Bit(FeatureIndex::kAVX512BF16) : 0;
235
0
    flags |= IsBitSet(abcd[3], 19) ? Bit(FeatureIndex::kAVX10) : 0;
236
0
    flags |= IsBitSet(abcd[3], 21) ? Bit(FeatureIndex::kAPX) : 0;
237
0
  }
238
239
0
  return flags;
240
0
}
241
242
// Each Highway target requires a 'group' of multiple features/flags.
243
static constexpr uint64_t kGroupSSE2 =
244
    Bit(FeatureIndex::kSSE) | Bit(FeatureIndex::kSSE2);
245
246
static constexpr uint64_t kGroupSSSE3 =
247
    Bit(FeatureIndex::kSSE3) | Bit(FeatureIndex::kSSSE3) | kGroupSSE2;
248
249
#ifdef HWY_DISABLE_PCLMUL_AES
250
static constexpr uint64_t kGroupSSE4 =
251
    Bit(FeatureIndex::kSSE41) | Bit(FeatureIndex::kSSE42) | kGroupSSSE3;
252
#else
253
static constexpr uint64_t kGroupSSE4 =
254
    Bit(FeatureIndex::kSSE41) | Bit(FeatureIndex::kSSE42) |
255
    Bit(FeatureIndex::kCLMUL) | Bit(FeatureIndex::kAES) | kGroupSSSE3;
256
#endif  // HWY_DISABLE_PCLMUL_AES
257
258
// We normally assume BMI/BMI2/FMA are available if AVX2 is. This allows us to
259
// use BZHI and (compiler-generated) MULX. However, VirtualBox lacks them
260
// [https://www.virtualbox.org/ticket/15471]. Thus we provide the option of
261
// avoiding using and requiring these so AVX2 can still be used.
262
#ifdef HWY_DISABLE_BMI2_FMA
263
static constexpr uint64_t kGroupBMI2_FMA = 0;
264
#else
265
static constexpr uint64_t kGroupBMI2_FMA = Bit(FeatureIndex::kBMI) |
266
                                           Bit(FeatureIndex::kBMI2) |
267
                                           Bit(FeatureIndex::kFMA);
268
#endif
269
270
#ifdef HWY_DISABLE_F16C
271
static constexpr uint64_t kGroupF16C = 0;
272
#else
273
static constexpr uint64_t kGroupF16C = Bit(FeatureIndex::kF16C);
274
#endif
275
276
static constexpr uint64_t kGroupAVX2 =
277
    Bit(FeatureIndex::kAVX) | Bit(FeatureIndex::kAVX2) |
278
    Bit(FeatureIndex::kLZCNT) | kGroupBMI2_FMA | kGroupF16C | kGroupSSE4;
279
280
static constexpr uint64_t kGroupAVX3 =
281
    Bit(FeatureIndex::kAVX512F) | Bit(FeatureIndex::kAVX512VL) |
282
    Bit(FeatureIndex::kAVX512DQ) | Bit(FeatureIndex::kAVX512BW) |
283
    Bit(FeatureIndex::kAVX512CD) | kGroupAVX2;
284
285
static constexpr uint64_t kGroupAVX3_DL =
286
    Bit(FeatureIndex::kVNNI) | Bit(FeatureIndex::kVPCLMULQDQ) |
287
    Bit(FeatureIndex::kVBMI) | Bit(FeatureIndex::kVBMI2) |
288
    Bit(FeatureIndex::kVAES) | Bit(FeatureIndex::kPOPCNTDQ) |
289
    Bit(FeatureIndex::kBITALG) | Bit(FeatureIndex::kGFNI) | kGroupAVX3;
290
291
static constexpr uint64_t kGroupAVX3_ZEN4 =
292
    Bit(FeatureIndex::kAVX512BF16) | kGroupAVX3_DL;
293
294
static constexpr uint64_t kGroupAVX3_SPR =
295
    Bit(FeatureIndex::kAVX512FP16) | kGroupAVX3_ZEN4;
296
297
static constexpr uint64_t kGroupAVX10 =
298
    Bit(FeatureIndex::kAVX10) | Bit(FeatureIndex::kAPX) |
299
    Bit(FeatureIndex::kVPCLMULQDQ) | Bit(FeatureIndex::kVAES) |
300
    Bit(FeatureIndex::kGFNI) | kGroupAVX2;
301
302
0
static int64_t DetectTargets() {
303
0
  int64_t bits = 0;  // return value of supported targets.
304
0
  HWY_IF_CONSTEXPR(HWY_ARCH_X86_64) {
305
0
    bits |= HWY_SSE2;  // always present in x64
306
0
  }
307
308
0
  const uint64_t flags = FlagsFromCPUID();
309
  // Set target bit(s) if all their group's flags are all set.
310
0
  if ((flags & kGroupAVX3_SPR) == kGroupAVX3_SPR) {
311
0
    bits |= HWY_AVX3_SPR;
312
0
  }
313
0
  if ((flags & kGroupAVX3_DL) == kGroupAVX3_DL) {
314
0
    bits |= HWY_AVX3_DL;
315
0
  }
316
0
  if ((flags & kGroupAVX3) == kGroupAVX3) {
317
0
    bits |= HWY_AVX3;
318
0
  }
319
0
  if ((flags & kGroupAVX2) == kGroupAVX2) {
320
0
    bits |= HWY_AVX2;
321
0
  }
322
0
  if ((flags & kGroupSSE4) == kGroupSSE4) {
323
0
    bits |= HWY_SSE4;
324
0
  }
325
0
  if ((flags & kGroupSSSE3) == kGroupSSSE3) {
326
0
    bits |= HWY_SSSE3;
327
0
  }
328
0
  HWY_IF_CONSTEXPR(HWY_ARCH_X86_32) {
329
    if ((flags & kGroupSSE2) == kGroupSSE2) {
330
      bits |= HWY_SSE2;
331
    }
332
  }
333
334
0
  uint32_t abcd[4];
335
336
0
  if ((flags & kGroupAVX10) == kGroupAVX10) {
337
0
    Cpuid(0x24, 0, abcd);
338
339
    // AVX10 version is in lower 8 bits of abcd[1]
340
0
    const uint32_t avx10_ver = abcd[1] & 0xFFu;
341
342
    // 512-bit vectors are supported if avx10_ver >= 1 is true and bit 18 of
343
    // abcd[1] is set
344
0
    const bool has_avx10_with_512bit_vectors =
345
0
        (avx10_ver >= 1) && IsBitSet(abcd[1], 18);
346
347
0
    if (has_avx10_with_512bit_vectors) {
348
      // AVX10.1 or later with support for 512-bit vectors implies support for
349
      // the AVX3/AVX3_DL/AVX3_SPR targets
350
0
      bits |= (HWY_AVX3_SPR | HWY_AVX3_DL | HWY_AVX3);
351
352
0
      if (avx10_ver >= 2) {
353
        // AVX10.2 is supported if avx10_ver >= 2 is true
354
0
        bits |= HWY_AVX10_2;
355
0
      }
356
0
    }
357
0
  }
358
359
  // Clear AVX2/AVX3 bits if the CPU or OS does not support XSAVE - otherwise,
360
  // YMM/ZMM registers are not preserved across context switches.
361
362
  // The lower 128 bits of XMM0-XMM15 are guaranteed to be preserved across
363
  // context switches on x86_64
364
365
  // The following OS's are known to preserve the lower 128 bits of XMM
366
  // registers across context switches on x86 CPUs that support SSE (even in
367
  // 32-bit mode):
368
  // - Windows 2000 or later
369
  // - Linux 2.4.0 or later
370
  // - Mac OS X 10.4 or later
371
  // - FreeBSD 4.4 or later
372
  // - NetBSD 1.6 or later
373
  // - OpenBSD 3.5 or later
374
  // - UnixWare 7 Release 7.1.1 or later
375
  // - Solaris 9 4/04 or later
376
377
0
  Cpuid(1, 0, abcd);
378
0
  const bool has_xsave = IsBitSet(abcd[2], 26);
379
0
  const bool has_osxsave = IsBitSet(abcd[2], 27);
380
0
  constexpr int64_t min_avx2 = HWY_AVX2 | (HWY_AVX2 - 1);
381
382
0
  if (has_xsave && has_osxsave) {
383
#if HWY_OS_APPLE
384
    // On macOS, check for AVX3 XSAVE support by checking that we are running on
385
    // macOS 12.2 or later and HasCpuFeature("hw.optional.avx512f") returns true
386
387
    // There is a bug in macOS 12.1 or earlier that can cause ZMM16-ZMM31, the
388
    // upper 256 bits of the ZMM registers, and K0-K7 (the AVX512 mask
389
    // registers) to not be properly preserved across a context switch on
390
    // macOS 12.1 or earlier.
391
392
    // This bug on macOS 12.1 or earlier on x86_64 CPU's with AVX3 support is
393
    // described at
394
    // https://community.intel.com/t5/Software-Tuning-Performance/MacOS-Darwin-kernel-bug-clobbers-AVX-512-opmask-register-state/m-p/1327259,
395
    // https://github.com/golang/go/issues/49233, and
396
    // https://github.com/simdutf/simdutf/pull/236.
397
398
    // In addition to the bug that is there on macOS 12.1 or earlier, bits 5, 6,
399
    // and 7 can be set to 0 on x86_64 CPUs with AVX3 support on macOS until
400
    // the first AVX512 instruction is executed as macOS only preserves
401
    // ZMM16-ZMM31, the upper 256 bits of the ZMM registers, and K0-K7 across a
402
    // context switch on threads that have executed an AVX512 instruction.
403
404
    // Checking for AVX3 XSAVE support on macOS using
405
    // HasCpuFeature("hw.optional.avx512f") avoids false negative results
406
    // on x86_64 CPU's that have AVX3 support.
407
    const bool have_avx3_xsave_support =
408
        IsMacOs12_2OrLater() && HasCpuFeature("hw.optional.avx512f");
409
#endif
410
411
0
    const uint32_t xcr0 = ReadXCR0();
412
0
    constexpr int64_t min_avx3 = HWY_AVX3 | (HWY_AVX3 - 1);
413
    // XMM/YMM
414
0
    if (!IsBitSet(xcr0, 1) || !IsBitSet(xcr0, 2)) {
415
      // Clear the AVX2/AVX3 bits if XMM/YMM XSAVE is not enabled
416
0
      bits &= ~min_avx2;
417
0
    }
418
419
0
#if !HWY_OS_APPLE
420
    // On OS's other than macOS, check for AVX3 XSAVE support by checking that
421
    // bits 5, 6, and 7 of XCR0 are set.
422
0
    const bool have_avx3_xsave_support =
423
0
        IsBitSet(xcr0, 5) && IsBitSet(xcr0, 6) && IsBitSet(xcr0, 7);
424
0
#endif
425
426
    // opmask, ZMM lo/hi
427
0
    if (!have_avx3_xsave_support) {
428
0
      bits &= ~min_avx3;
429
0
    }
430
0
  } else {  // !has_xsave || !has_osxsave
431
    // Clear the AVX2/AVX3 bits if the CPU or OS does not support XSAVE
432
0
    bits &= ~min_avx2;
433
0
  }
434
435
  // This is mainly to work around the slow Zen4 CompressStore. It's unclear
436
  // whether subsequent AMD models will be affected; assume yes.
437
0
  if ((bits & HWY_AVX3_DL) && (flags & kGroupAVX3_ZEN4) == kGroupAVX3_ZEN4 &&
438
0
      IsAMD()) {
439
0
    bits |= HWY_AVX3_ZEN4;
440
0
  }
441
442
0
  return bits;
443
0
}
444
445
}  // namespace x86
446
#elif HWY_ARCH_ARM && HWY_HAVE_RUNTIME_DISPATCH
447
namespace arm {
448
449
#if HWY_ARCH_ARM_A64 && !HWY_OS_APPLE &&        \
450
    (HWY_COMPILER_GCC || HWY_COMPILER_CLANG) && \
451
    ((HWY_TARGETS & HWY_ALL_SVE) != 0)
452
HWY_PUSH_ATTRIBUTES("+sve")
453
static int64_t DetectAdditionalSveTargets(int64_t detected_targets) {
454
  uint64_t sve_vec_len;
455
456
  // Use inline assembly instead of svcntb_pat(SV_ALL) as GCC or Clang might
457
  // possibly optimize a svcntb_pat(SV_ALL) call to a constant if the
458
  // -msve-vector-bits option is specified
459
  asm("cntb %0" : "=r"(sve_vec_len)::);
460
461
  return ((sve_vec_len == 32)
462
              ? HWY_SVE_256
463
              : (((detected_targets & HWY_SVE2) != 0 && sve_vec_len == 16)
464
                     ? HWY_SVE2_128
465
                     : 0));
466
}
467
HWY_POP_ATTRIBUTES
468
#endif
469
470
static int64_t DetectTargets() {
471
  int64_t bits = 0;  // return value of supported targets.
472
473
  using CapBits = unsigned long;  // NOLINT
474
#if HWY_OS_APPLE
475
  const CapBits hw = 0UL;
476
#else
477
  // For Android, this has been supported since API 20 (2014).
478
  const CapBits hw = getauxval(AT_HWCAP);
479
#endif
480
  (void)hw;
481
482
#if HWY_ARCH_ARM_A64
483
  bits |= HWY_NEON_WITHOUT_AES;  // aarch64 always has NEON and VFPv4..
484
485
#if HWY_OS_APPLE
486
  if (HasCpuFeature("hw.optional.arm.FEAT_AES")) {
487
    bits |= HWY_NEON;
488
489
    // Some macOS versions report AdvSIMD_HPFPCvt under a different key.
490
    // Check both known variants for compatibility.
491
    if ((HasCpuFeature("hw.optional.AdvSIMD_HPFPCvt") ||
492
         HasCpuFeature("hw.optional.arm.AdvSIMD_HPFPCvt")) &&
493
        HasCpuFeature("hw.optional.arm.FEAT_DotProd") &&
494
        HasCpuFeature("hw.optional.arm.FEAT_BF16") &&
495
        HasCpuFeature("hw.optional.arm.FEAT_I8MM")) {
496
      bits |= HWY_NEON_BF16;
497
    }
498
  }
499
#else  // !HWY_OS_APPLE
500
  // .. but not necessarily AES, which is required for HWY_NEON.
501
#if defined(HWCAP_AES)
502
  if (hw & HWCAP_AES) {
503
    bits |= HWY_NEON;
504
505
#if defined(HWCAP_ASIMDHP) && defined(HWCAP_ASIMDDP) && defined(HWCAP2_BF16)
506
    const CapBits hw2 = getauxval(AT_HWCAP2);
507
    constexpr CapBits kGroupF16Dot = HWCAP_ASIMDHP | HWCAP_ASIMDDP;
508
    constexpr CapBits kGroupBF16 = HWCAP2_BF16;
509
    if ((hw & kGroupF16Dot) == kGroupF16Dot &&
510
        (hw2 & kGroupBF16) == kGroupBF16) {
511
      bits |= HWY_NEON_BF16;
512
    }
513
#endif  // HWCAP_ASIMDHP && HWCAP_ASIMDDP && HWCAP2_BF16
514
  }
515
#endif  // HWCAP_AES
516
517
#if defined(HWCAP_SVE)
518
  if (hw & HWCAP_SVE) {
519
    bits |= HWY_SVE;
520
  }
521
#endif
522
523
#ifndef HWCAP2_SVE2
524
#define HWCAP2_SVE2 (1 << 1)
525
#endif
526
#ifndef HWCAP2_SVEAES
527
#define HWCAP2_SVEAES (1 << 2)
528
#endif
529
#ifndef HWCAP2_SVEI8MM
530
#define HWCAP2_SVEI8MM (1 << 9)
531
#endif
532
#ifndef HWCAP2_SVEBF16
533
#define HWCAP2_SVEBF16 (1 << 12)
534
#endif
535
536
  constexpr CapBits kGroupSVE2 = HWCAP2_SVE2 | HWCAP2_SVEAES;
537
  const CapBits hw2 = getauxval(AT_HWCAP2);
538
  if ((hw2 & kGroupSVE2) == kGroupSVE2) {
539
    bits |= HWY_SVE2;
540
  }
541
542
#if (HWY_COMPILER_GCC || HWY_COMPILER_CLANG) && \
543
    ((HWY_TARGETS & HWY_ALL_SVE) != 0)
544
  if ((bits & HWY_ALL_SVE) != 0) {
545
    bits |= DetectAdditionalSveTargets(bits);
546
547
    // SVE2_128 implies I8MM and BF16, hence remove it if they are not present.
548
    constexpr CapBits kGroupSVE2_128 = HWCAP2_SVEI8MM | HWCAP2_SVEBF16;
549
    if ((hw2 & kGroupSVE2_128) != kGroupSVE2_128) {
550
      bits &= ~HWY_SVE2_128;
551
    }
552
  }
553
#endif  // (HWY_COMPILER_GCC || HWY_COMPILER_CLANG) &&
554
        // ((HWY_TARGETS & HWY_ALL_SVE) != 0)
555
556
#endif  // HWY_OS_APPLE
557
558
#else  // !HWY_ARCH_ARM_A64
559
560
// Some old auxv.h / hwcap.h do not define these. If not, treat as unsupported.
561
#if defined(HWCAP_NEON) && defined(HWCAP_VFPv4)
562
  if ((hw & HWCAP_NEON) && (hw & HWCAP_VFPv4)) {
563
    bits |= HWY_NEON_WITHOUT_AES;
564
  }
565
#endif
566
567
  // aarch32 would check getauxval(AT_HWCAP2) & HWCAP2_AES, but we do not yet
568
  // support that platform, and Armv7 lacks AES entirely. Because HWY_NEON
569
  // requires native AES instructions, we do not enable that target here.
570
571
#endif  // HWY_ARCH_ARM_A64
572
  return bits;
573
}
574
}  // namespace arm
575
#elif HWY_ARCH_PPC && HWY_HAVE_RUNTIME_DISPATCH
576
namespace ppc {
577
578
#ifndef PPC_FEATURE_HAS_ALTIVEC
579
#define PPC_FEATURE_HAS_ALTIVEC 0x10000000
580
#endif
581
582
#ifndef PPC_FEATURE_HAS_VSX
583
#define PPC_FEATURE_HAS_VSX 0x00000080
584
#endif
585
586
#ifndef PPC_FEATURE2_ARCH_2_07
587
#define PPC_FEATURE2_ARCH_2_07 0x80000000
588
#endif
589
590
#ifndef PPC_FEATURE2_VEC_CRYPTO
591
#define PPC_FEATURE2_VEC_CRYPTO 0x02000000
592
#endif
593
594
#ifndef PPC_FEATURE2_ARCH_3_00
595
#define PPC_FEATURE2_ARCH_3_00 0x00800000
596
#endif
597
598
#ifndef PPC_FEATURE2_ARCH_3_1
599
#define PPC_FEATURE2_ARCH_3_1 0x00040000
600
#endif
601
602
using CapBits = unsigned long;  // NOLINT
603
604
// For AT_HWCAP, the others are for AT_HWCAP2
605
static constexpr CapBits kGroupVSX =
606
    PPC_FEATURE_HAS_ALTIVEC | PPC_FEATURE_HAS_VSX;
607
608
#if defined(HWY_DISABLE_PPC8_CRYPTO)
609
static constexpr CapBits kGroupPPC8 = PPC_FEATURE2_ARCH_2_07;
610
#else
611
static constexpr CapBits kGroupPPC8 =
612
    PPC_FEATURE2_ARCH_2_07 | PPC_FEATURE2_VEC_CRYPTO;
613
#endif
614
static constexpr CapBits kGroupPPC9 = kGroupPPC8 | PPC_FEATURE2_ARCH_3_00;
615
static constexpr CapBits kGroupPPC10 = kGroupPPC9 | PPC_FEATURE2_ARCH_3_1;
616
617
static int64_t DetectTargets() {
618
  int64_t bits = 0;  // return value of supported targets.
619
620
#if defined(AT_HWCAP) && defined(AT_HWCAP2)
621
  const CapBits hw = getauxval(AT_HWCAP);
622
623
  if ((hw & kGroupVSX) == kGroupVSX) {
624
    const CapBits hw2 = getauxval(AT_HWCAP2);
625
    if ((hw2 & kGroupPPC8) == kGroupPPC8) {
626
      bits |= HWY_PPC8;
627
    }
628
    if ((hw2 & kGroupPPC9) == kGroupPPC9) {
629
      bits |= HWY_PPC9;
630
    }
631
    if ((hw2 & kGroupPPC10) == kGroupPPC10) {
632
      bits |= HWY_PPC10;
633
    }
634
  }  // VSX
635
#endif  // defined(AT_HWCAP) && defined(AT_HWCAP2)
636
637
  return bits;
638
}
639
}  // namespace ppc
640
#elif HWY_ARCH_S390X && HWY_HAVE_RUNTIME_DISPATCH
641
namespace s390x {
642
643
#ifndef HWCAP_S390_VX
644
#define HWCAP_S390_VX 2048
645
#endif
646
647
#ifndef HWCAP_S390_VXE
648
#define HWCAP_S390_VXE 8192
649
#endif
650
651
#ifndef HWCAP_S390_VXRS_EXT2
652
#define HWCAP_S390_VXRS_EXT2 32768
653
#endif
654
655
using CapBits = unsigned long;  // NOLINT
656
657
static constexpr CapBits kGroupZ14 = HWCAP_S390_VX | HWCAP_S390_VXE;
658
static constexpr CapBits kGroupZ15 =
659
    HWCAP_S390_VX | HWCAP_S390_VXE | HWCAP_S390_VXRS_EXT2;
660
661
static int64_t DetectTargets() {
662
  int64_t bits = 0;
663
664
#if defined(AT_HWCAP)
665
  const CapBits hw = getauxval(AT_HWCAP);
666
667
  if ((hw & kGroupZ14) == kGroupZ14) {
668
    bits |= HWY_Z14;
669
  }
670
671
  if ((hw & kGroupZ15) == kGroupZ15) {
672
    bits |= HWY_Z15;
673
  }
674
#endif
675
676
  return bits;
677
}
678
}  // namespace s390x
679
#elif HWY_ARCH_RISCV && HWY_HAVE_RUNTIME_DISPATCH
680
namespace rvv {
681
682
#ifndef HWCAP_RVV
683
#define COMPAT_HWCAP_ISA_V (1 << ('V' - 'A'))
684
#endif
685
686
using CapBits = unsigned long;  // NOLINT
687
688
static int64_t DetectTargets() {
689
  int64_t bits = 0;
690
691
  const CapBits hw = getauxval(AT_HWCAP);
692
693
  if ((hw & COMPAT_HWCAP_ISA_V) == COMPAT_HWCAP_ISA_V) {
694
    size_t e8m1_vec_len;
695
#if HWY_ARCH_RISCV_64
696
    int64_t vtype_reg_val;
697
#else
698
    int32_t vtype_reg_val;
699
#endif
700
701
    // Check that a vuint8m1_t vector is at least 16 bytes and that tail
702
    // agnostic and mask agnostic mode are supported
703
    asm volatile(
704
        // Avoid compiler error on GCC or Clang if -march=rv64gcv1p0 or
705
        // -march=rv32gcv1p0 option is not specified on the command line
706
        ".option push\n\t"
707
        ".option arch, +v\n\t"
708
        "vsetvli %0, zero, e8, m1, ta, ma\n\t"
709
        "csrr %1, vtype\n\t"
710
        ".option pop"
711
        : "=r"(e8m1_vec_len), "=r"(vtype_reg_val));
712
713
    // The RVV target is supported if the VILL bit of VTYPE (the MSB bit of
714
    // VTYPE) is not set and the length of a vuint8m1_t vector is at least 16
715
    // bytes
716
    if (vtype_reg_val >= 0 && e8m1_vec_len >= 16) {
717
      bits |= HWY_RVV;
718
    }
719
  }
720
721
  return bits;
722
}
723
}  // namespace rvv
724
#elif HWY_ARCH_LOONGARCH && HWY_HAVE_RUNTIME_DISPATCH
725
726
namespace loongarch {
727
728
#ifndef LA_HWCAP_LSX
729
#define LA_HWCAP_LSX (1u << 4)
730
#endif
731
#ifndef LA_HWCAP_LASX
732
#define LA_HWCAP_LASX (1u << 5)
733
#endif
734
735
using CapBits = unsigned long;  // NOLINT
736
737
static int64_t DetectTargets() {
738
  int64_t bits = 0;
739
  const CapBits hw = getauxval(AT_HWCAP);
740
  if (hw & LA_HWCAP_LSX) bits |= HWY_LSX;
741
  if (hw & LA_HWCAP_LASX) bits |= HWY_LASX;
742
  return bits;
743
}
744
}  // namespace loongarch
745
#endif  // HWY_ARCH_*
746
747
// Returns targets supported by the CPU, independently of DisableTargets.
748
// Factored out of SupportedTargets to make its structure more obvious. Note
749
// that x86 CPUID may take several hundred cycles.
750
0
static int64_t DetectTargets() {
751
  // Apps will use only one of these (the default is EMU128), but compile flags
752
  // for this TU may differ from that of the app, so allow both.
753
0
  int64_t bits = HWY_SCALAR | HWY_EMU128;
754
755
0
#if HWY_ARCH_X86 && HWY_HAVE_RUNTIME_DISPATCH
756
0
  bits |= x86::DetectTargets();
757
#elif HWY_ARCH_ARM && HWY_HAVE_RUNTIME_DISPATCH
758
  bits |= arm::DetectTargets();
759
#elif HWY_ARCH_PPC && HWY_HAVE_RUNTIME_DISPATCH
760
  bits |= ppc::DetectTargets();
761
#elif HWY_ARCH_S390X && HWY_HAVE_RUNTIME_DISPATCH
762
  bits |= s390x::DetectTargets();
763
#elif HWY_ARCH_RISCV && HWY_HAVE_RUNTIME_DISPATCH
764
  bits |= rvv::DetectTargets();
765
#elif HWY_ARCH_LOONGARCH && HWY_HAVE_RUNTIME_DISPATCH
766
  bits |= loongarch::DetectTargets();
767
768
#else
769
  // TODO(janwas): detect support for WASM.
770
  // This file is typically compiled without HWY_IS_TEST, but targets_test has
771
  // it set, and will expect all of its HWY_TARGETS (= all attainable) to be
772
  // supported.
773
  bits |= HWY_ENABLED_BASELINE;
774
#endif  // HWY_ARCH_*
775
776
0
  if ((bits & HWY_ENABLED_BASELINE) != HWY_ENABLED_BASELINE) {
777
0
    const uint64_t bits_u = static_cast<uint64_t>(bits);
778
0
    const uint64_t enabled = static_cast<uint64_t>(HWY_ENABLED_BASELINE);
779
0
    HWY_WARN("CPU supports 0x%08x%08x, software requires 0x%08x%08x\n",
780
0
             static_cast<uint32_t>(bits_u >> 32),
781
0
             static_cast<uint32_t>(bits_u & 0xFFFFFFFF),
782
0
             static_cast<uint32_t>(enabled >> 32),
783
0
             static_cast<uint32_t>(enabled & 0xFFFFFFFF));
784
0
  }
785
786
0
  return bits;
787
0
}
788
789
// When running tests, this value can be set to the mocked supported targets
790
// mask. Only written to from a single thread before the test starts.
791
static int64_t supported_targets_for_test_ = 0;
792
793
// Mask of targets disabled at runtime with DisableTargets.
794
static int64_t supported_mask_ = LimitsMax<int64_t>();
795
796
0
HWY_DLLEXPORT void DisableTargets(int64_t disabled_targets) {
797
0
  supported_mask_ = static_cast<int64_t>(~disabled_targets);
798
  // This will take effect on the next call to SupportedTargets, which is
799
  // called right before GetChosenTarget::Update. However, calling Update here
800
  // would make it appear that HWY_DYNAMIC_DISPATCH was called, which we want
801
  // to check in tests. We instead de-initialize such that the next
802
  // HWY_DYNAMIC_DISPATCH calls GetChosenTarget::Update via FunctionCache.
803
0
  GetChosenTarget().DeInit();
804
0
}
805
806
0
HWY_DLLEXPORT void SetSupportedTargetsForTest(int64_t targets) {
807
0
  supported_targets_for_test_ = targets;
808
0
  GetChosenTarget().DeInit();  // see comment above
809
0
}
810
811
0
HWY_DLLEXPORT int64_t SupportedTargets() {
812
0
  int64_t targets = supported_targets_for_test_;
813
0
  if (HWY_LIKELY(targets == 0)) {
814
    // Mock not active. Re-detect instead of caching just in case we're on a
815
    // heterogeneous ISA (also requires some app support to pin threads). This
816
    // is only reached on the first HWY_DYNAMIC_DISPATCH or after each call to
817
    // DisableTargets or SetSupportedTargetsForTest.
818
0
    targets = DetectTargets();
819
820
    // VectorBytes invokes HWY_DYNAMIC_DISPATCH. To prevent infinite recursion,
821
    // first set up ChosenTarget. No need to Update() again afterwards with the
822
    // final targets - that will be done by a caller of this function.
823
0
    GetChosenTarget().Update(targets);
824
0
  }
825
826
0
  targets &= supported_mask_;
827
0
  return targets == 0 ? HWY_STATIC_TARGET : targets;
828
0
}
829
830
0
HWY_DLLEXPORT ChosenTarget& GetChosenTarget() {
831
0
  static ChosenTarget chosen_target;
832
0
  return chosen_target;
833
0
}
834
835
}  // namespace hwy