/work/workdir/UnpackedTarball/highway/hwy/targets.cc
Line | Count | Source |
1 | | // Copyright 2019 Google LLC |
2 | | // Copyright 2025 Arm Limited and/or its affiliates <open-source-office@arm.com> |
3 | | // SPDX-License-Identifier: Apache-2.0 |
4 | | // |
5 | | // Licensed under the Apache License, Version 2.0 (the "License"); |
6 | | // you may not use this file except in compliance with the License. |
7 | | // You may obtain a copy of the License at |
8 | | // |
9 | | // http://www.apache.org/licenses/LICENSE-2.0 |
10 | | // |
11 | | // Unless required by applicable law or agreed to in writing, software |
12 | | // distributed under the License is distributed on an "AS IS" BASIS, |
13 | | // WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. |
14 | | // See the License for the specific language governing permissions and |
15 | | // limitations under the License. |
16 | | |
17 | | #include "hwy/targets.h" |
18 | | |
19 | | #include <stdint.h> |
20 | | #include <stdio.h> |
21 | | |
22 | | #include "hwy/base.h" |
23 | | #include "hwy/detect_targets.h" |
24 | | #include "hwy/highway.h" |
25 | | #include "hwy/x86_cpuid.h" |
26 | | |
27 | | #if HWY_ARCH_X86 |
28 | | #include <xmmintrin.h> |
29 | | |
30 | | #elif (HWY_ARCH_ARM || HWY_ARCH_PPC || HWY_ARCH_S390X || HWY_ARCH_RISCV || \ |
31 | | HWY_ARCH_LOONGARCH) && \ |
32 | | HWY_OS_LINUX |
33 | | // sys/auxv.h does not always include asm/hwcap.h, or define HWCAP*, hence we |
34 | | // still include this directly. See #1199. |
35 | | #if HWY_HAVE_ASM_HWCAP |
36 | | #include <asm/hwcap.h> |
37 | | #endif |
38 | | #if HWY_HAVE_AUXV |
39 | | #include <sys/auxv.h> |
40 | | #endif |
41 | | |
42 | | #endif // HWY_ARCH_* |
43 | | |
44 | | #if HWY_OS_APPLE |
45 | | #include <sys/sysctl.h> |
46 | | #include <sys/utsname.h> |
47 | | #endif // HWY_OS_APPLE |
48 | | |
49 | | namespace hwy { |
50 | | |
51 | | #if HWY_OS_APPLE |
52 | | static HWY_INLINE HWY_MAYBE_UNUSED bool HasCpuFeature( |
53 | | const char* feature_name) { |
54 | | int result = 0; |
55 | | size_t len = sizeof(int); |
56 | | return (sysctlbyname(feature_name, &result, &len, nullptr, 0) == 0 && |
57 | | result != 0); |
58 | | } |
59 | | |
60 | | static HWY_INLINE HWY_MAYBE_UNUSED bool ParseU32(const char*& ptr, |
61 | | uint32_t& parsed_val) { |
62 | | uint64_t parsed_u64 = 0; |
63 | | |
64 | | const char* start_ptr = ptr; |
65 | | for (char ch; (ch = (*ptr)) != '\0'; ++ptr) { |
66 | | unsigned digit = static_cast<unsigned>(static_cast<unsigned char>(ch)) - |
67 | | static_cast<unsigned>(static_cast<unsigned char>('0')); |
68 | | if (digit > 9u) { |
69 | | break; |
70 | | } |
71 | | |
72 | | parsed_u64 = (parsed_u64 * 10u) + digit; |
73 | | if (parsed_u64 > 0xFFFFFFFFu) { |
74 | | return false; |
75 | | } |
76 | | } |
77 | | |
78 | | parsed_val = static_cast<uint32_t>(parsed_u64); |
79 | | return (ptr != start_ptr); |
80 | | } |
81 | | |
82 | | static HWY_INLINE HWY_MAYBE_UNUSED bool IsMacOs12_2OrLater() { |
83 | | utsname uname_buf; |
84 | | ZeroBytes(&uname_buf, sizeof(utsname)); |
85 | | |
86 | | if ((uname(&uname_buf)) != 0) { |
87 | | return false; |
88 | | } |
89 | | |
90 | | const char* ptr = uname_buf.release; |
91 | | if (!ptr) { |
92 | | return false; |
93 | | } |
94 | | |
95 | | uint32_t major; |
96 | | uint32_t minor; |
97 | | if (!ParseU32(ptr, major)) { |
98 | | return false; |
99 | | } |
100 | | |
101 | | if (*ptr != '.') { |
102 | | return false; |
103 | | } |
104 | | |
105 | | ++ptr; |
106 | | if (!ParseU32(ptr, minor)) { |
107 | | return false; |
108 | | } |
109 | | |
110 | | // We are running on macOS 12.2 or later if the Darwin kernel version is 21.3 |
111 | | // or later |
112 | | return (major > 21 || (major == 21 && minor >= 3)); |
113 | | } |
114 | | #endif // HWY_OS_APPLE |
115 | | |
116 | | #if HWY_ARCH_X86 && HWY_HAVE_RUNTIME_DISPATCH |
117 | | namespace x86 { |
118 | | |
119 | | // Returns the lower 32 bits of extended control register 0. |
120 | | // Requires CPU support for "OSXSAVE" (see below). |
121 | 0 | static uint32_t ReadXCR0() { |
122 | | #if HWY_COMPILER_MSVC |
123 | | return static_cast<uint32_t>(_xgetbv(0)); |
124 | | #else // HWY_COMPILER_MSVC |
125 | 0 | uint32_t xcr0, xcr0_high; |
126 | 0 | const uint32_t index = 0; |
127 | 0 | asm volatile(".byte 0x0F, 0x01, 0xD0" |
128 | 0 | : "=a"(xcr0), "=d"(xcr0_high) |
129 | 0 | : "c"(index)); |
130 | 0 | return xcr0; |
131 | 0 | #endif // HWY_COMPILER_MSVC |
132 | 0 | } |
133 | | |
134 | | // Arbitrary bit indices indicating which instruction set extensions are |
135 | | // supported. Use enum to ensure values are distinct. |
136 | | enum class FeatureIndex : uint32_t { |
137 | | kSSE = 0, |
138 | | kSSE2, |
139 | | kSSE3, |
140 | | kSSSE3, |
141 | | |
142 | | kSSE41, |
143 | | kSSE42, |
144 | | kCLMUL, |
145 | | kAES, |
146 | | |
147 | | kAVX, |
148 | | kAVX2, |
149 | | kF16C, |
150 | | kFMA, |
151 | | kLZCNT, |
152 | | kBMI, |
153 | | kBMI2, |
154 | | |
155 | | kAVX512F, |
156 | | kAVX512VL, |
157 | | kAVX512CD, |
158 | | kAVX512DQ, |
159 | | kAVX512BW, |
160 | | kAVX512FP16, |
161 | | kAVX512BF16, |
162 | | |
163 | | kVNNI, |
164 | | kVPCLMULQDQ, |
165 | | kVBMI, |
166 | | kVBMI2, |
167 | | kVAES, |
168 | | kPOPCNTDQ, |
169 | | kBITALG, |
170 | | kGFNI, |
171 | | |
172 | | kAVX10, |
173 | | kAPX, |
174 | | |
175 | | kSentinel |
176 | | }; |
177 | | static_assert(static_cast<size_t>(FeatureIndex::kSentinel) < 64, |
178 | | "Too many bits for u64"); |
179 | | |
180 | 0 | static HWY_INLINE constexpr uint64_t Bit(FeatureIndex index) { |
181 | 0 | return 1ull << static_cast<size_t>(index); |
182 | 0 | } |
183 | | |
184 | | // Returns bit array of FeatureIndex from CPUID feature flags. |
185 | 0 | static uint64_t FlagsFromCPUID() { |
186 | 0 | uint64_t flags = 0; // return value |
187 | 0 | uint32_t abcd[4]; |
188 | 0 | Cpuid(0, 0, abcd); |
189 | 0 | const uint32_t max_level = abcd[0]; |
190 | | |
191 | | // Standard feature flags |
192 | 0 | Cpuid(1, 0, abcd); |
193 | 0 | flags |= IsBitSet(abcd[3], 25) ? Bit(FeatureIndex::kSSE) : 0; |
194 | 0 | flags |= IsBitSet(abcd[3], 26) ? Bit(FeatureIndex::kSSE2) : 0; |
195 | 0 | flags |= IsBitSet(abcd[2], 0) ? Bit(FeatureIndex::kSSE3) : 0; |
196 | 0 | flags |= IsBitSet(abcd[2], 1) ? Bit(FeatureIndex::kCLMUL) : 0; |
197 | 0 | flags |= IsBitSet(abcd[2], 9) ? Bit(FeatureIndex::kSSSE3) : 0; |
198 | 0 | flags |= IsBitSet(abcd[2], 12) ? Bit(FeatureIndex::kFMA) : 0; |
199 | 0 | flags |= IsBitSet(abcd[2], 19) ? Bit(FeatureIndex::kSSE41) : 0; |
200 | 0 | flags |= IsBitSet(abcd[2], 20) ? Bit(FeatureIndex::kSSE42) : 0; |
201 | 0 | flags |= IsBitSet(abcd[2], 25) ? Bit(FeatureIndex::kAES) : 0; |
202 | 0 | flags |= IsBitSet(abcd[2], 28) ? Bit(FeatureIndex::kAVX) : 0; |
203 | 0 | flags |= IsBitSet(abcd[2], 29) ? Bit(FeatureIndex::kF16C) : 0; |
204 | | |
205 | | // Extended feature flags |
206 | 0 | Cpuid(0x80000001U, 0, abcd); |
207 | 0 | flags |= IsBitSet(abcd[2], 5) ? Bit(FeatureIndex::kLZCNT) : 0; |
208 | | |
209 | | // Extended features |
210 | 0 | if (max_level >= 7) { |
211 | 0 | Cpuid(7, 0, abcd); |
212 | 0 | flags |= IsBitSet(abcd[1], 3) ? Bit(FeatureIndex::kBMI) : 0; |
213 | 0 | flags |= IsBitSet(abcd[1], 5) ? Bit(FeatureIndex::kAVX2) : 0; |
214 | 0 | flags |= IsBitSet(abcd[1], 8) ? Bit(FeatureIndex::kBMI2) : 0; |
215 | |
|
216 | 0 | flags |= IsBitSet(abcd[1], 16) ? Bit(FeatureIndex::kAVX512F) : 0; |
217 | 0 | flags |= IsBitSet(abcd[1], 17) ? Bit(FeatureIndex::kAVX512DQ) : 0; |
218 | 0 | flags |= IsBitSet(abcd[1], 28) ? Bit(FeatureIndex::kAVX512CD) : 0; |
219 | 0 | flags |= IsBitSet(abcd[1], 30) ? Bit(FeatureIndex::kAVX512BW) : 0; |
220 | 0 | flags |= IsBitSet(abcd[1], 31) ? Bit(FeatureIndex::kAVX512VL) : 0; |
221 | |
|
222 | 0 | flags |= IsBitSet(abcd[2], 1) ? Bit(FeatureIndex::kVBMI) : 0; |
223 | 0 | flags |= IsBitSet(abcd[2], 6) ? Bit(FeatureIndex::kVBMI2) : 0; |
224 | 0 | flags |= IsBitSet(abcd[2], 8) ? Bit(FeatureIndex::kGFNI) : 0; |
225 | 0 | flags |= IsBitSet(abcd[2], 9) ? Bit(FeatureIndex::kVAES) : 0; |
226 | 0 | flags |= IsBitSet(abcd[2], 10) ? Bit(FeatureIndex::kVPCLMULQDQ) : 0; |
227 | 0 | flags |= IsBitSet(abcd[2], 11) ? Bit(FeatureIndex::kVNNI) : 0; |
228 | 0 | flags |= IsBitSet(abcd[2], 12) ? Bit(FeatureIndex::kBITALG) : 0; |
229 | 0 | flags |= IsBitSet(abcd[2], 14) ? Bit(FeatureIndex::kPOPCNTDQ) : 0; |
230 | |
|
231 | 0 | flags |= IsBitSet(abcd[3], 23) ? Bit(FeatureIndex::kAVX512FP16) : 0; |
232 | |
|
233 | 0 | Cpuid(7, 1, abcd); |
234 | 0 | flags |= IsBitSet(abcd[0], 5) ? Bit(FeatureIndex::kAVX512BF16) : 0; |
235 | 0 | flags |= IsBitSet(abcd[3], 19) ? Bit(FeatureIndex::kAVX10) : 0; |
236 | 0 | flags |= IsBitSet(abcd[3], 21) ? Bit(FeatureIndex::kAPX) : 0; |
237 | 0 | } |
238 | |
|
239 | 0 | return flags; |
240 | 0 | } |
241 | | |
242 | | // Each Highway target requires a 'group' of multiple features/flags. |
243 | | static constexpr uint64_t kGroupSSE2 = |
244 | | Bit(FeatureIndex::kSSE) | Bit(FeatureIndex::kSSE2); |
245 | | |
246 | | static constexpr uint64_t kGroupSSSE3 = |
247 | | Bit(FeatureIndex::kSSE3) | Bit(FeatureIndex::kSSSE3) | kGroupSSE2; |
248 | | |
249 | | #ifdef HWY_DISABLE_PCLMUL_AES |
250 | | static constexpr uint64_t kGroupSSE4 = |
251 | | Bit(FeatureIndex::kSSE41) | Bit(FeatureIndex::kSSE42) | kGroupSSSE3; |
252 | | #else |
253 | | static constexpr uint64_t kGroupSSE4 = |
254 | | Bit(FeatureIndex::kSSE41) | Bit(FeatureIndex::kSSE42) | |
255 | | Bit(FeatureIndex::kCLMUL) | Bit(FeatureIndex::kAES) | kGroupSSSE3; |
256 | | #endif // HWY_DISABLE_PCLMUL_AES |
257 | | |
258 | | // We normally assume BMI/BMI2/FMA are available if AVX2 is. This allows us to |
259 | | // use BZHI and (compiler-generated) MULX. However, VirtualBox lacks them |
260 | | // [https://www.virtualbox.org/ticket/15471]. Thus we provide the option of |
261 | | // avoiding using and requiring these so AVX2 can still be used. |
262 | | #ifdef HWY_DISABLE_BMI2_FMA |
263 | | static constexpr uint64_t kGroupBMI2_FMA = 0; |
264 | | #else |
265 | | static constexpr uint64_t kGroupBMI2_FMA = Bit(FeatureIndex::kBMI) | |
266 | | Bit(FeatureIndex::kBMI2) | |
267 | | Bit(FeatureIndex::kFMA); |
268 | | #endif |
269 | | |
270 | | #ifdef HWY_DISABLE_F16C |
271 | | static constexpr uint64_t kGroupF16C = 0; |
272 | | #else |
273 | | static constexpr uint64_t kGroupF16C = Bit(FeatureIndex::kF16C); |
274 | | #endif |
275 | | |
276 | | static constexpr uint64_t kGroupAVX2 = |
277 | | Bit(FeatureIndex::kAVX) | Bit(FeatureIndex::kAVX2) | |
278 | | Bit(FeatureIndex::kLZCNT) | kGroupBMI2_FMA | kGroupF16C | kGroupSSE4; |
279 | | |
280 | | static constexpr uint64_t kGroupAVX3 = |
281 | | Bit(FeatureIndex::kAVX512F) | Bit(FeatureIndex::kAVX512VL) | |
282 | | Bit(FeatureIndex::kAVX512DQ) | Bit(FeatureIndex::kAVX512BW) | |
283 | | Bit(FeatureIndex::kAVX512CD) | kGroupAVX2; |
284 | | |
285 | | static constexpr uint64_t kGroupAVX3_DL = |
286 | | Bit(FeatureIndex::kVNNI) | Bit(FeatureIndex::kVPCLMULQDQ) | |
287 | | Bit(FeatureIndex::kVBMI) | Bit(FeatureIndex::kVBMI2) | |
288 | | Bit(FeatureIndex::kVAES) | Bit(FeatureIndex::kPOPCNTDQ) | |
289 | | Bit(FeatureIndex::kBITALG) | Bit(FeatureIndex::kGFNI) | kGroupAVX3; |
290 | | |
291 | | static constexpr uint64_t kGroupAVX3_ZEN4 = |
292 | | Bit(FeatureIndex::kAVX512BF16) | kGroupAVX3_DL; |
293 | | |
294 | | static constexpr uint64_t kGroupAVX3_SPR = |
295 | | Bit(FeatureIndex::kAVX512FP16) | kGroupAVX3_ZEN4; |
296 | | |
297 | | static constexpr uint64_t kGroupAVX10 = |
298 | | Bit(FeatureIndex::kAVX10) | Bit(FeatureIndex::kAPX) | |
299 | | Bit(FeatureIndex::kVPCLMULQDQ) | Bit(FeatureIndex::kVAES) | |
300 | | Bit(FeatureIndex::kGFNI) | kGroupAVX2; |
301 | | |
302 | 0 | static int64_t DetectTargets() { |
303 | 0 | int64_t bits = 0; // return value of supported targets. |
304 | 0 | HWY_IF_CONSTEXPR(HWY_ARCH_X86_64) { |
305 | 0 | bits |= HWY_SSE2; // always present in x64 |
306 | 0 | } |
307 | |
|
308 | 0 | const uint64_t flags = FlagsFromCPUID(); |
309 | | // Set target bit(s) if all their group's flags are all set. |
310 | 0 | if ((flags & kGroupAVX3_SPR) == kGroupAVX3_SPR) { |
311 | 0 | bits |= HWY_AVX3_SPR; |
312 | 0 | } |
313 | 0 | if ((flags & kGroupAVX3_DL) == kGroupAVX3_DL) { |
314 | 0 | bits |= HWY_AVX3_DL; |
315 | 0 | } |
316 | 0 | if ((flags & kGroupAVX3) == kGroupAVX3) { |
317 | 0 | bits |= HWY_AVX3; |
318 | 0 | } |
319 | 0 | if ((flags & kGroupAVX2) == kGroupAVX2) { |
320 | 0 | bits |= HWY_AVX2; |
321 | 0 | } |
322 | 0 | if ((flags & kGroupSSE4) == kGroupSSE4) { |
323 | 0 | bits |= HWY_SSE4; |
324 | 0 | } |
325 | 0 | if ((flags & kGroupSSSE3) == kGroupSSSE3) { |
326 | 0 | bits |= HWY_SSSE3; |
327 | 0 | } |
328 | 0 | HWY_IF_CONSTEXPR(HWY_ARCH_X86_32) { |
329 | | if ((flags & kGroupSSE2) == kGroupSSE2) { |
330 | | bits |= HWY_SSE2; |
331 | | } |
332 | | } |
333 | |
|
334 | 0 | uint32_t abcd[4]; |
335 | |
|
336 | 0 | if ((flags & kGroupAVX10) == kGroupAVX10) { |
337 | 0 | Cpuid(0x24, 0, abcd); |
338 | | |
339 | | // AVX10 version is in lower 8 bits of abcd[1] |
340 | 0 | const uint32_t avx10_ver = abcd[1] & 0xFFu; |
341 | | |
342 | | // 512-bit vectors are supported if avx10_ver >= 1 is true and bit 18 of |
343 | | // abcd[1] is set |
344 | 0 | const bool has_avx10_with_512bit_vectors = |
345 | 0 | (avx10_ver >= 1) && IsBitSet(abcd[1], 18); |
346 | |
|
347 | 0 | if (has_avx10_with_512bit_vectors) { |
348 | | // AVX10.1 or later with support for 512-bit vectors implies support for |
349 | | // the AVX3/AVX3_DL/AVX3_SPR targets |
350 | 0 | bits |= (HWY_AVX3_SPR | HWY_AVX3_DL | HWY_AVX3); |
351 | |
|
352 | 0 | if (avx10_ver >= 2) { |
353 | | // AVX10.2 is supported if avx10_ver >= 2 is true |
354 | 0 | bits |= HWY_AVX10_2; |
355 | 0 | } |
356 | 0 | } |
357 | 0 | } |
358 | | |
359 | | // Clear AVX2/AVX3 bits if the CPU or OS does not support XSAVE - otherwise, |
360 | | // YMM/ZMM registers are not preserved across context switches. |
361 | | |
362 | | // The lower 128 bits of XMM0-XMM15 are guaranteed to be preserved across |
363 | | // context switches on x86_64 |
364 | | |
365 | | // The following OS's are known to preserve the lower 128 bits of XMM |
366 | | // registers across context switches on x86 CPUs that support SSE (even in |
367 | | // 32-bit mode): |
368 | | // - Windows 2000 or later |
369 | | // - Linux 2.4.0 or later |
370 | | // - Mac OS X 10.4 or later |
371 | | // - FreeBSD 4.4 or later |
372 | | // - NetBSD 1.6 or later |
373 | | // - OpenBSD 3.5 or later |
374 | | // - UnixWare 7 Release 7.1.1 or later |
375 | | // - Solaris 9 4/04 or later |
376 | |
|
377 | 0 | Cpuid(1, 0, abcd); |
378 | 0 | const bool has_xsave = IsBitSet(abcd[2], 26); |
379 | 0 | const bool has_osxsave = IsBitSet(abcd[2], 27); |
380 | 0 | constexpr int64_t min_avx2 = HWY_AVX2 | (HWY_AVX2 - 1); |
381 | |
|
382 | 0 | if (has_xsave && has_osxsave) { |
383 | | #if HWY_OS_APPLE |
384 | | // On macOS, check for AVX3 XSAVE support by checking that we are running on |
385 | | // macOS 12.2 or later and HasCpuFeature("hw.optional.avx512f") returns true |
386 | | |
387 | | // There is a bug in macOS 12.1 or earlier that can cause ZMM16-ZMM31, the |
388 | | // upper 256 bits of the ZMM registers, and K0-K7 (the AVX512 mask |
389 | | // registers) to not be properly preserved across a context switch on |
390 | | // macOS 12.1 or earlier. |
391 | | |
392 | | // This bug on macOS 12.1 or earlier on x86_64 CPU's with AVX3 support is |
393 | | // described at |
394 | | // https://community.intel.com/t5/Software-Tuning-Performance/MacOS-Darwin-kernel-bug-clobbers-AVX-512-opmask-register-state/m-p/1327259, |
395 | | // https://github.com/golang/go/issues/49233, and |
396 | | // https://github.com/simdutf/simdutf/pull/236. |
397 | | |
398 | | // In addition to the bug that is there on macOS 12.1 or earlier, bits 5, 6, |
399 | | // and 7 can be set to 0 on x86_64 CPUs with AVX3 support on macOS until |
400 | | // the first AVX512 instruction is executed as macOS only preserves |
401 | | // ZMM16-ZMM31, the upper 256 bits of the ZMM registers, and K0-K7 across a |
402 | | // context switch on threads that have executed an AVX512 instruction. |
403 | | |
404 | | // Checking for AVX3 XSAVE support on macOS using |
405 | | // HasCpuFeature("hw.optional.avx512f") avoids false negative results |
406 | | // on x86_64 CPU's that have AVX3 support. |
407 | | const bool have_avx3_xsave_support = |
408 | | IsMacOs12_2OrLater() && HasCpuFeature("hw.optional.avx512f"); |
409 | | #endif |
410 | |
|
411 | 0 | const uint32_t xcr0 = ReadXCR0(); |
412 | 0 | constexpr int64_t min_avx3 = HWY_AVX3 | (HWY_AVX3 - 1); |
413 | | // XMM/YMM |
414 | 0 | if (!IsBitSet(xcr0, 1) || !IsBitSet(xcr0, 2)) { |
415 | | // Clear the AVX2/AVX3 bits if XMM/YMM XSAVE is not enabled |
416 | 0 | bits &= ~min_avx2; |
417 | 0 | } |
418 | |
|
419 | 0 | #if !HWY_OS_APPLE |
420 | | // On OS's other than macOS, check for AVX3 XSAVE support by checking that |
421 | | // bits 5, 6, and 7 of XCR0 are set. |
422 | 0 | const bool have_avx3_xsave_support = |
423 | 0 | IsBitSet(xcr0, 5) && IsBitSet(xcr0, 6) && IsBitSet(xcr0, 7); |
424 | 0 | #endif |
425 | | |
426 | | // opmask, ZMM lo/hi |
427 | 0 | if (!have_avx3_xsave_support) { |
428 | 0 | bits &= ~min_avx3; |
429 | 0 | } |
430 | 0 | } else { // !has_xsave || !has_osxsave |
431 | | // Clear the AVX2/AVX3 bits if the CPU or OS does not support XSAVE |
432 | 0 | bits &= ~min_avx2; |
433 | 0 | } |
434 | | |
435 | | // This is mainly to work around the slow Zen4 CompressStore. It's unclear |
436 | | // whether subsequent AMD models will be affected; assume yes. |
437 | 0 | if ((bits & HWY_AVX3_DL) && (flags & kGroupAVX3_ZEN4) == kGroupAVX3_ZEN4 && |
438 | 0 | IsAMD()) { |
439 | 0 | bits |= HWY_AVX3_ZEN4; |
440 | 0 | } |
441 | |
|
442 | 0 | return bits; |
443 | 0 | } |
444 | | |
445 | | } // namespace x86 |
446 | | #elif HWY_ARCH_ARM && HWY_HAVE_RUNTIME_DISPATCH |
447 | | namespace arm { |
448 | | |
449 | | #if HWY_ARCH_ARM_A64 && !HWY_OS_APPLE && \ |
450 | | (HWY_COMPILER_GCC || HWY_COMPILER_CLANG) && \ |
451 | | ((HWY_TARGETS & HWY_ALL_SVE) != 0) |
452 | | HWY_PUSH_ATTRIBUTES("+sve") |
453 | | static int64_t DetectAdditionalSveTargets(int64_t detected_targets) { |
454 | | uint64_t sve_vec_len; |
455 | | |
456 | | // Use inline assembly instead of svcntb_pat(SV_ALL) as GCC or Clang might |
457 | | // possibly optimize a svcntb_pat(SV_ALL) call to a constant if the |
458 | | // -msve-vector-bits option is specified |
459 | | asm("cntb %0" : "=r"(sve_vec_len)::); |
460 | | |
461 | | return ((sve_vec_len == 32) |
462 | | ? HWY_SVE_256 |
463 | | : (((detected_targets & HWY_SVE2) != 0 && sve_vec_len == 16) |
464 | | ? HWY_SVE2_128 |
465 | | : 0)); |
466 | | } |
467 | | HWY_POP_ATTRIBUTES |
468 | | #endif |
469 | | |
470 | | static int64_t DetectTargets() { |
471 | | int64_t bits = 0; // return value of supported targets. |
472 | | |
473 | | using CapBits = unsigned long; // NOLINT |
474 | | #if HWY_OS_APPLE |
475 | | const CapBits hw = 0UL; |
476 | | #else |
477 | | // For Android, this has been supported since API 20 (2014). |
478 | | const CapBits hw = getauxval(AT_HWCAP); |
479 | | #endif |
480 | | (void)hw; |
481 | | |
482 | | #if HWY_ARCH_ARM_A64 |
483 | | bits |= HWY_NEON_WITHOUT_AES; // aarch64 always has NEON and VFPv4.. |
484 | | |
485 | | #if HWY_OS_APPLE |
486 | | if (HasCpuFeature("hw.optional.arm.FEAT_AES")) { |
487 | | bits |= HWY_NEON; |
488 | | |
489 | | // Some macOS versions report AdvSIMD_HPFPCvt under a different key. |
490 | | // Check both known variants for compatibility. |
491 | | if ((HasCpuFeature("hw.optional.AdvSIMD_HPFPCvt") || |
492 | | HasCpuFeature("hw.optional.arm.AdvSIMD_HPFPCvt")) && |
493 | | HasCpuFeature("hw.optional.arm.FEAT_DotProd") && |
494 | | HasCpuFeature("hw.optional.arm.FEAT_BF16") && |
495 | | HasCpuFeature("hw.optional.arm.FEAT_I8MM")) { |
496 | | bits |= HWY_NEON_BF16; |
497 | | } |
498 | | } |
499 | | #else // !HWY_OS_APPLE |
500 | | // .. but not necessarily AES, which is required for HWY_NEON. |
501 | | #if defined(HWCAP_AES) |
502 | | if (hw & HWCAP_AES) { |
503 | | bits |= HWY_NEON; |
504 | | |
505 | | #if defined(HWCAP_ASIMDHP) && defined(HWCAP_ASIMDDP) && defined(HWCAP2_BF16) |
506 | | const CapBits hw2 = getauxval(AT_HWCAP2); |
507 | | constexpr CapBits kGroupF16Dot = HWCAP_ASIMDHP | HWCAP_ASIMDDP; |
508 | | constexpr CapBits kGroupBF16 = HWCAP2_BF16; |
509 | | if ((hw & kGroupF16Dot) == kGroupF16Dot && |
510 | | (hw2 & kGroupBF16) == kGroupBF16) { |
511 | | bits |= HWY_NEON_BF16; |
512 | | } |
513 | | #endif // HWCAP_ASIMDHP && HWCAP_ASIMDDP && HWCAP2_BF16 |
514 | | } |
515 | | #endif // HWCAP_AES |
516 | | |
517 | | #if defined(HWCAP_SVE) |
518 | | if (hw & HWCAP_SVE) { |
519 | | bits |= HWY_SVE; |
520 | | } |
521 | | #endif |
522 | | |
523 | | #ifndef HWCAP2_SVE2 |
524 | | #define HWCAP2_SVE2 (1 << 1) |
525 | | #endif |
526 | | #ifndef HWCAP2_SVEAES |
527 | | #define HWCAP2_SVEAES (1 << 2) |
528 | | #endif |
529 | | #ifndef HWCAP2_SVEI8MM |
530 | | #define HWCAP2_SVEI8MM (1 << 9) |
531 | | #endif |
532 | | #ifndef HWCAP2_SVEBF16 |
533 | | #define HWCAP2_SVEBF16 (1 << 12) |
534 | | #endif |
535 | | |
536 | | constexpr CapBits kGroupSVE2 = HWCAP2_SVE2 | HWCAP2_SVEAES; |
537 | | const CapBits hw2 = getauxval(AT_HWCAP2); |
538 | | if ((hw2 & kGroupSVE2) == kGroupSVE2) { |
539 | | bits |= HWY_SVE2; |
540 | | } |
541 | | |
542 | | #if (HWY_COMPILER_GCC || HWY_COMPILER_CLANG) && \ |
543 | | ((HWY_TARGETS & HWY_ALL_SVE) != 0) |
544 | | if ((bits & HWY_ALL_SVE) != 0) { |
545 | | bits |= DetectAdditionalSveTargets(bits); |
546 | | |
547 | | // SVE2_128 implies I8MM and BF16, hence remove it if they are not present. |
548 | | constexpr CapBits kGroupSVE2_128 = HWCAP2_SVEI8MM | HWCAP2_SVEBF16; |
549 | | if ((hw2 & kGroupSVE2_128) != kGroupSVE2_128) { |
550 | | bits &= ~HWY_SVE2_128; |
551 | | } |
552 | | } |
553 | | #endif // (HWY_COMPILER_GCC || HWY_COMPILER_CLANG) && |
554 | | // ((HWY_TARGETS & HWY_ALL_SVE) != 0) |
555 | | |
556 | | #endif // HWY_OS_APPLE |
557 | | |
558 | | #else // !HWY_ARCH_ARM_A64 |
559 | | |
560 | | // Some old auxv.h / hwcap.h do not define these. If not, treat as unsupported. |
561 | | #if defined(HWCAP_NEON) && defined(HWCAP_VFPv4) |
562 | | if ((hw & HWCAP_NEON) && (hw & HWCAP_VFPv4)) { |
563 | | bits |= HWY_NEON_WITHOUT_AES; |
564 | | } |
565 | | #endif |
566 | | |
567 | | // aarch32 would check getauxval(AT_HWCAP2) & HWCAP2_AES, but we do not yet |
568 | | // support that platform, and Armv7 lacks AES entirely. Because HWY_NEON |
569 | | // requires native AES instructions, we do not enable that target here. |
570 | | |
571 | | #endif // HWY_ARCH_ARM_A64 |
572 | | return bits; |
573 | | } |
574 | | } // namespace arm |
575 | | #elif HWY_ARCH_PPC && HWY_HAVE_RUNTIME_DISPATCH |
576 | | namespace ppc { |
577 | | |
578 | | #ifndef PPC_FEATURE_HAS_ALTIVEC |
579 | | #define PPC_FEATURE_HAS_ALTIVEC 0x10000000 |
580 | | #endif |
581 | | |
582 | | #ifndef PPC_FEATURE_HAS_VSX |
583 | | #define PPC_FEATURE_HAS_VSX 0x00000080 |
584 | | #endif |
585 | | |
586 | | #ifndef PPC_FEATURE2_ARCH_2_07 |
587 | | #define PPC_FEATURE2_ARCH_2_07 0x80000000 |
588 | | #endif |
589 | | |
590 | | #ifndef PPC_FEATURE2_VEC_CRYPTO |
591 | | #define PPC_FEATURE2_VEC_CRYPTO 0x02000000 |
592 | | #endif |
593 | | |
594 | | #ifndef PPC_FEATURE2_ARCH_3_00 |
595 | | #define PPC_FEATURE2_ARCH_3_00 0x00800000 |
596 | | #endif |
597 | | |
598 | | #ifndef PPC_FEATURE2_ARCH_3_1 |
599 | | #define PPC_FEATURE2_ARCH_3_1 0x00040000 |
600 | | #endif |
601 | | |
602 | | using CapBits = unsigned long; // NOLINT |
603 | | |
604 | | // For AT_HWCAP, the others are for AT_HWCAP2 |
605 | | static constexpr CapBits kGroupVSX = |
606 | | PPC_FEATURE_HAS_ALTIVEC | PPC_FEATURE_HAS_VSX; |
607 | | |
608 | | #if defined(HWY_DISABLE_PPC8_CRYPTO) |
609 | | static constexpr CapBits kGroupPPC8 = PPC_FEATURE2_ARCH_2_07; |
610 | | #else |
611 | | static constexpr CapBits kGroupPPC8 = |
612 | | PPC_FEATURE2_ARCH_2_07 | PPC_FEATURE2_VEC_CRYPTO; |
613 | | #endif |
614 | | static constexpr CapBits kGroupPPC9 = kGroupPPC8 | PPC_FEATURE2_ARCH_3_00; |
615 | | static constexpr CapBits kGroupPPC10 = kGroupPPC9 | PPC_FEATURE2_ARCH_3_1; |
616 | | |
617 | | static int64_t DetectTargets() { |
618 | | int64_t bits = 0; // return value of supported targets. |
619 | | |
620 | | #if defined(AT_HWCAP) && defined(AT_HWCAP2) |
621 | | const CapBits hw = getauxval(AT_HWCAP); |
622 | | |
623 | | if ((hw & kGroupVSX) == kGroupVSX) { |
624 | | const CapBits hw2 = getauxval(AT_HWCAP2); |
625 | | if ((hw2 & kGroupPPC8) == kGroupPPC8) { |
626 | | bits |= HWY_PPC8; |
627 | | } |
628 | | if ((hw2 & kGroupPPC9) == kGroupPPC9) { |
629 | | bits |= HWY_PPC9; |
630 | | } |
631 | | if ((hw2 & kGroupPPC10) == kGroupPPC10) { |
632 | | bits |= HWY_PPC10; |
633 | | } |
634 | | } // VSX |
635 | | #endif // defined(AT_HWCAP) && defined(AT_HWCAP2) |
636 | | |
637 | | return bits; |
638 | | } |
639 | | } // namespace ppc |
640 | | #elif HWY_ARCH_S390X && HWY_HAVE_RUNTIME_DISPATCH |
641 | | namespace s390x { |
642 | | |
643 | | #ifndef HWCAP_S390_VX |
644 | | #define HWCAP_S390_VX 2048 |
645 | | #endif |
646 | | |
647 | | #ifndef HWCAP_S390_VXE |
648 | | #define HWCAP_S390_VXE 8192 |
649 | | #endif |
650 | | |
651 | | #ifndef HWCAP_S390_VXRS_EXT2 |
652 | | #define HWCAP_S390_VXRS_EXT2 32768 |
653 | | #endif |
654 | | |
655 | | using CapBits = unsigned long; // NOLINT |
656 | | |
657 | | static constexpr CapBits kGroupZ14 = HWCAP_S390_VX | HWCAP_S390_VXE; |
658 | | static constexpr CapBits kGroupZ15 = |
659 | | HWCAP_S390_VX | HWCAP_S390_VXE | HWCAP_S390_VXRS_EXT2; |
660 | | |
661 | | static int64_t DetectTargets() { |
662 | | int64_t bits = 0; |
663 | | |
664 | | #if defined(AT_HWCAP) |
665 | | const CapBits hw = getauxval(AT_HWCAP); |
666 | | |
667 | | if ((hw & kGroupZ14) == kGroupZ14) { |
668 | | bits |= HWY_Z14; |
669 | | } |
670 | | |
671 | | if ((hw & kGroupZ15) == kGroupZ15) { |
672 | | bits |= HWY_Z15; |
673 | | } |
674 | | #endif |
675 | | |
676 | | return bits; |
677 | | } |
678 | | } // namespace s390x |
679 | | #elif HWY_ARCH_RISCV && HWY_HAVE_RUNTIME_DISPATCH |
680 | | namespace rvv { |
681 | | |
682 | | #ifndef HWCAP_RVV |
683 | | #define COMPAT_HWCAP_ISA_V (1 << ('V' - 'A')) |
684 | | #endif |
685 | | |
686 | | using CapBits = unsigned long; // NOLINT |
687 | | |
688 | | static int64_t DetectTargets() { |
689 | | int64_t bits = 0; |
690 | | |
691 | | const CapBits hw = getauxval(AT_HWCAP); |
692 | | |
693 | | if ((hw & COMPAT_HWCAP_ISA_V) == COMPAT_HWCAP_ISA_V) { |
694 | | size_t e8m1_vec_len; |
695 | | #if HWY_ARCH_RISCV_64 |
696 | | int64_t vtype_reg_val; |
697 | | #else |
698 | | int32_t vtype_reg_val; |
699 | | #endif |
700 | | |
701 | | // Check that a vuint8m1_t vector is at least 16 bytes and that tail |
702 | | // agnostic and mask agnostic mode are supported |
703 | | asm volatile( |
704 | | // Avoid compiler error on GCC or Clang if -march=rv64gcv1p0 or |
705 | | // -march=rv32gcv1p0 option is not specified on the command line |
706 | | ".option push\n\t" |
707 | | ".option arch, +v\n\t" |
708 | | "vsetvli %0, zero, e8, m1, ta, ma\n\t" |
709 | | "csrr %1, vtype\n\t" |
710 | | ".option pop" |
711 | | : "=r"(e8m1_vec_len), "=r"(vtype_reg_val)); |
712 | | |
713 | | // The RVV target is supported if the VILL bit of VTYPE (the MSB bit of |
714 | | // VTYPE) is not set and the length of a vuint8m1_t vector is at least 16 |
715 | | // bytes |
716 | | if (vtype_reg_val >= 0 && e8m1_vec_len >= 16) { |
717 | | bits |= HWY_RVV; |
718 | | } |
719 | | } |
720 | | |
721 | | return bits; |
722 | | } |
723 | | } // namespace rvv |
724 | | #elif HWY_ARCH_LOONGARCH && HWY_HAVE_RUNTIME_DISPATCH |
725 | | |
726 | | namespace loongarch { |
727 | | |
728 | | #ifndef LA_HWCAP_LSX |
729 | | #define LA_HWCAP_LSX (1u << 4) |
730 | | #endif |
731 | | #ifndef LA_HWCAP_LASX |
732 | | #define LA_HWCAP_LASX (1u << 5) |
733 | | #endif |
734 | | |
735 | | using CapBits = unsigned long; // NOLINT |
736 | | |
737 | | static int64_t DetectTargets() { |
738 | | int64_t bits = 0; |
739 | | const CapBits hw = getauxval(AT_HWCAP); |
740 | | if (hw & LA_HWCAP_LSX) bits |= HWY_LSX; |
741 | | if (hw & LA_HWCAP_LASX) bits |= HWY_LASX; |
742 | | return bits; |
743 | | } |
744 | | } // namespace loongarch |
745 | | #endif // HWY_ARCH_* |
746 | | |
747 | | // Returns targets supported by the CPU, independently of DisableTargets. |
748 | | // Factored out of SupportedTargets to make its structure more obvious. Note |
749 | | // that x86 CPUID may take several hundred cycles. |
750 | 0 | static int64_t DetectTargets() { |
751 | | // Apps will use only one of these (the default is EMU128), but compile flags |
752 | | // for this TU may differ from that of the app, so allow both. |
753 | 0 | int64_t bits = HWY_SCALAR | HWY_EMU128; |
754 | |
|
755 | 0 | #if HWY_ARCH_X86 && HWY_HAVE_RUNTIME_DISPATCH |
756 | 0 | bits |= x86::DetectTargets(); |
757 | | #elif HWY_ARCH_ARM && HWY_HAVE_RUNTIME_DISPATCH |
758 | | bits |= arm::DetectTargets(); |
759 | | #elif HWY_ARCH_PPC && HWY_HAVE_RUNTIME_DISPATCH |
760 | | bits |= ppc::DetectTargets(); |
761 | | #elif HWY_ARCH_S390X && HWY_HAVE_RUNTIME_DISPATCH |
762 | | bits |= s390x::DetectTargets(); |
763 | | #elif HWY_ARCH_RISCV && HWY_HAVE_RUNTIME_DISPATCH |
764 | | bits |= rvv::DetectTargets(); |
765 | | #elif HWY_ARCH_LOONGARCH && HWY_HAVE_RUNTIME_DISPATCH |
766 | | bits |= loongarch::DetectTargets(); |
767 | | |
768 | | #else |
769 | | // TODO(janwas): detect support for WASM. |
770 | | // This file is typically compiled without HWY_IS_TEST, but targets_test has |
771 | | // it set, and will expect all of its HWY_TARGETS (= all attainable) to be |
772 | | // supported. |
773 | | bits |= HWY_ENABLED_BASELINE; |
774 | | #endif // HWY_ARCH_* |
775 | |
|
776 | 0 | if ((bits & HWY_ENABLED_BASELINE) != HWY_ENABLED_BASELINE) { |
777 | 0 | const uint64_t bits_u = static_cast<uint64_t>(bits); |
778 | 0 | const uint64_t enabled = static_cast<uint64_t>(HWY_ENABLED_BASELINE); |
779 | 0 | HWY_WARN("CPU supports 0x%08x%08x, software requires 0x%08x%08x\n", |
780 | 0 | static_cast<uint32_t>(bits_u >> 32), |
781 | 0 | static_cast<uint32_t>(bits_u & 0xFFFFFFFF), |
782 | 0 | static_cast<uint32_t>(enabled >> 32), |
783 | 0 | static_cast<uint32_t>(enabled & 0xFFFFFFFF)); |
784 | 0 | } |
785 | |
|
786 | 0 | return bits; |
787 | 0 | } |
788 | | |
789 | | // When running tests, this value can be set to the mocked supported targets |
790 | | // mask. Only written to from a single thread before the test starts. |
791 | | static int64_t supported_targets_for_test_ = 0; |
792 | | |
793 | | // Mask of targets disabled at runtime with DisableTargets. |
794 | | static int64_t supported_mask_ = LimitsMax<int64_t>(); |
795 | | |
796 | 0 | HWY_DLLEXPORT void DisableTargets(int64_t disabled_targets) { |
797 | 0 | supported_mask_ = static_cast<int64_t>(~disabled_targets); |
798 | | // This will take effect on the next call to SupportedTargets, which is |
799 | | // called right before GetChosenTarget::Update. However, calling Update here |
800 | | // would make it appear that HWY_DYNAMIC_DISPATCH was called, which we want |
801 | | // to check in tests. We instead de-initialize such that the next |
802 | | // HWY_DYNAMIC_DISPATCH calls GetChosenTarget::Update via FunctionCache. |
803 | 0 | GetChosenTarget().DeInit(); |
804 | 0 | } |
805 | | |
806 | 0 | HWY_DLLEXPORT void SetSupportedTargetsForTest(int64_t targets) { |
807 | 0 | supported_targets_for_test_ = targets; |
808 | 0 | GetChosenTarget().DeInit(); // see comment above |
809 | 0 | } |
810 | | |
811 | 0 | HWY_DLLEXPORT int64_t SupportedTargets() { |
812 | 0 | int64_t targets = supported_targets_for_test_; |
813 | 0 | if (HWY_LIKELY(targets == 0)) { |
814 | | // Mock not active. Re-detect instead of caching just in case we're on a |
815 | | // heterogeneous ISA (also requires some app support to pin threads). This |
816 | | // is only reached on the first HWY_DYNAMIC_DISPATCH or after each call to |
817 | | // DisableTargets or SetSupportedTargetsForTest. |
818 | 0 | targets = DetectTargets(); |
819 | | |
820 | | // VectorBytes invokes HWY_DYNAMIC_DISPATCH. To prevent infinite recursion, |
821 | | // first set up ChosenTarget. No need to Update() again afterwards with the |
822 | | // final targets - that will be done by a caller of this function. |
823 | 0 | GetChosenTarget().Update(targets); |
824 | 0 | } |
825 | |
|
826 | 0 | targets &= supported_mask_; |
827 | 0 | return targets == 0 ? HWY_STATIC_TARGET : targets; |
828 | 0 | } |
829 | | |
830 | 0 | HWY_DLLEXPORT ChosenTarget& GetChosenTarget() { |
831 | 0 | static ChosenTarget chosen_target; |
832 | 0 | return chosen_target; |
833 | 0 | } |
834 | | |
835 | | } // namespace hwy |