libstdc++
simd_details.h
1// Implementation of <simd> -*- C++ -*-
2
3// Copyright The GNU Toolchain Authors.
4//
5// This file is part of the GNU ISO C++ Library. This library is free
6// software; you can redistribute it and/or modify it under the
7// terms of the GNU General Public License as published by the
8// Free Software Foundation; either version 3, or (at your option)
9// any later version.
10
11// This library is distributed in the hope that it will be useful,
12// but WITHOUT ANY WARRANTY; without even the implied warranty of
13// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
14// GNU General Public License for more details.
15
16// Under Section 7 of GPL version 3, you are granted additional
17// permissions described in the GCC Runtime Library Exception, version
18// 3.1, as published by the Free Software Foundation.
19
20// You should have received a copy of the GNU General Public License and
21// a copy of the GCC Runtime Library Exception along with this program;
22// see the files COPYING3 and COPYING.RUNTIME respectively. If not, see
23// <http://www.gnu.org/licenses/>.
24
25#ifndef _GLIBCXX_SIMD_DETAILS_H
26#define _GLIBCXX_SIMD_DETAILS_H 1
27
28#ifdef _GLIBCXX_SYSHDR
29#pragma GCC system_header
30#endif
31
32#if __cplusplus >= 202400L
33
34#include <bit>
35#include <bits/c++config.h> // _GLIBCXX_FLOAT_IS_IEEE_BINARY32
36#include <bits/stl_function.h> // plus, minus, multiplies, ...
37#include <bits/utility.h> // integer_sequence, etc.
38#include <cmath> // for math_errhandling :(
39#include <concepts>
40#include <cstdint>
41#include <limits>
42#include <span> // for dynamic_extent
43
44#if __CHAR_BIT__ != 8
45// There are simply too many constants and bit operators that currently depend on CHAR_BIT == 8.
46// Generalization to CHAR_BIT != 8 does not make sense without testability (i.e. a test target).
47#error "<simd> is not supported for CHAR_BIT != 8"
48#endif
49
50// psabi warnings are bogus because the ABI of the internal types never leaks into user code
51#pragma GCC diagnostic push
52#pragma GCC diagnostic ignored "-Wpsabi"
53
54#if defined __x86_64__ || defined __i386__
55#define _GLIBCXX_X86 1
56#else
57#define _GLIBCXX_X86 0
58#endif
59
60#ifndef _GLIBCXX_SIMD_NOEXCEPT
61/** @internal
62 * For unit-testing preconditions, use this macro to remove noexcept.
63 */
64#define _GLIBCXX_SIMD_NOEXCEPT noexcept
65#endif
66
67#define _GLIBCXX_SIMD_TOSTRING_IMPL(x) #x
68#define _GLIBCXX_SIMD_TOSTRING(x) _GLIBCXX_SIMD_TOSTRING_IMPL(x)
69
70// This is used for unit-testing precondition checking
71#define __glibcxx_simd_precondition(expr, msg, ...) \
72 __glibcxx_assert(expr)
73
74namespace std _GLIBCXX_VISIBILITY(default)
75{
76_GLIBCXX_BEGIN_NAMESPACE_VERSION
77
78 template<typename> class complex;
79
80namespace simd
81{
82 template <typename _Tp>
83 inline constexpr _Tp
84 __iota = [] { static_assert(false, "invalid __iota specialization"); }();
85
86 // [simd.general] vectorizable types
87 template <typename _Tp>
88 concept __complex_like_impl
89 = same_as<_Tp, complex<typename _Tp::value_type>>;
90
91 /** @internal
92 * Satisfied if @p _Tp implements the std::complex interface.
93 */
94 template <typename _Tp>
95 concept __complex_like = __complex_like_impl<remove_cvref_t<_Tp>>;
96
97 template <typename _Tp>
98 concept __vectorizable_scalar
99 = same_as<remove_cv_t<_Tp>, _Tp>
100#ifdef __STDCPP_BFLOAT16_T__
101 && !same_as<_Tp, __gnu_cxx::__bfloat16_t>
102#endif
103 && ((integral<_Tp> && sizeof(_Tp) <= sizeof(0ULL) && !same_as<_Tp, bool>)
104 || (floating_point<_Tp> && sizeof(_Tp) <= sizeof(double)));
105
106 // [simd.general] p2
107 template <typename _Tp>
108 concept __vectorizable
109 = __vectorizable_scalar<_Tp>
110 || (__complex_like_impl<_Tp> && __vectorizable_scalar<typename _Tp::value_type>
111 && floating_point<typename _Tp::value_type>);
112
113 /** @internal
114 * Describes variants of _Abi.
115 */
116 enum class _AbiVariant : unsigned long long
117 {
118 _BitMask = 0x01, // AVX512 bit-masks
119 _MaskVariants = 0x0f, // vector masks if bits [0:3] are 0
120 _CxIleav = 0x10, // store complex components interleaved (ririri...)
121 // mask elements are stored for both (001122...)
122 _CxCtgus = 0x20, // ... or store complex components contiguously (rrrr iiii)
123 // mask elements are store for one component (0123)
124 _CxVariants = _CxIleav | _CxCtgus,
125 };
126
127 /** @internal
128 * Return @p __in with only bits set that are set in any of @p __to_keep.
129 */
130 consteval _AbiVariant
131 __filter_abi_variant(_AbiVariant __in, same_as<_AbiVariant> auto... __to_keep)
132 {
134 return static_cast<_AbiVariant>(static_cast<_Up>(__in) & (static_cast<_Up>(__to_keep) | ...));
135 }
136
137 /** @internal
138 * Type used whenever no valid integer/value type exists.
139 */
140 struct _InvalidInteger
141 {};
142
143 /** @internal
144 * Alias for a signed integer type T such that sizeof(T) equals _Bytes.
145 *
146 * C++26 [simd.expos.defn]
147 */
148 template <size_t _Bytes>
149 using __integer_from
150 = decltype([] consteval {
151 if constexpr (sizeof(signed char) == _Bytes)
152 return static_cast<signed char>(0);
153 else if constexpr (sizeof(signed short) == _Bytes)
154 return static_cast<signed short>(0);
155 else if constexpr (sizeof(signed int) == _Bytes)
156 return static_cast<signed int>(0);
157 else if constexpr (sizeof(signed long long) == _Bytes)
158 return static_cast<signed long long>(0);
159 else
160 return _InvalidInteger();
161 }());
162
163 template <size_t _Bytes>
164 using __float_from = decltype([] consteval {
165 if constexpr (sizeof(double) == _Bytes)
166 return double();
167 else if constexpr (sizeof(float) == _Bytes)
168 return float();
169 else if constexpr (sizeof(_Float16) == _Bytes)
170 return _Float16();
171 }());
172
173 /** @internal
174 * Alias for an unsigned integer type T such that sizeof(T) equals _Bytes.
175 */
176 template <size_t _Bytes>
178
179 /** @internal
180 * Divide @p __x by @p __y while rounding up instead of down.
181 *
182 * Preconditions: __x >= 0 && __y > 0.
183 */
184 template <typename _Tp>
185 consteval _Tp
186 __div_ceil(_Tp __x, _Tp __y)
187 { return (__x + __y - 1) / __y; }
188
189 /** @internal
190 * Alias for an unsigned integer type that can store at least @p _NBits bits.
191 */
192 template <int _NBits>
193 requires (_NBits > 0 && _NBits <= numeric_limits<unsigned long long>::digits)
194 using _Bitmask = _UInt<__div_ceil(__bit_ceil(unsigned(_NBits)), unsigned(__CHAR_BIT__))>;
195
196 /** @internal
197 * Map a given type @p _Tp to an equivalent type.
198 *
199 * This helps with reducing the necessary branches && casts in the implementation as well as
200 * reducing the number of template instantiations.
201 */
202 template <typename _Tp>
203 struct __canonical_vec_type
204 { using type = _Tp; };
205
206 template <typename _Tp>
207 using __canonical_vec_type_t = typename __canonical_vec_type<_Tp>::type;
208
209#if __SIZEOF_INT__ == __SIZEOF_LONG__
210 template <>
211 struct __canonical_vec_type<long>
212 { using type = int; };
213
214 template <>
215 struct __canonical_vec_type<unsigned long>
216 { using type = unsigned int; };
217#elif __SIZEOF_LONG_LONG__ == __SIZEOF_LONG__
218 template <>
219 struct __canonical_vec_type<long>
220 { using type = long long; };
221
222 template <>
223 struct __canonical_vec_type<unsigned long>
224 { using type = unsigned long long; };
225#endif
226
227 template <typename _Tp>
228 requires std::is_enum_v<_Tp>
229 struct __canonical_vec_type<_Tp>
230 { using type = __canonical_vec_type<std::underlying_type_t<_Tp>>::type; };
231
232 template <>
233 struct __canonical_vec_type<char>
234#if __CHAR_UNSIGNED__
235 { using type = unsigned char; };
236#else
237 { using type = signed char; };
238#endif
239
240 template <>
241 struct __canonical_vec_type<char8_t>
242 { using type = unsigned char; };
243
244 template <>
245 struct __canonical_vec_type<char16_t>
246 { using type = uint_least16_t; };
247
248 template <>
249 struct __canonical_vec_type<char32_t>
250 { using type = uint_least32_t; };
251
252 template <>
253 struct __canonical_vec_type<wchar_t>
254 {
255 using type = std::__conditional_t<std::is_signed_v<wchar_t>,
256 simd::__integer_from<sizeof(wchar_t)>,
257 simd::_UInt<sizeof(wchar_t)>>;
258 };
259
260#if defined(__FLT64_DIG__) && defined(_GLIBCXX_DOUBLE_IS_IEEE_BINARY64)
261 template <>
262 struct __canonical_vec_type<_Float64>
263 { using type = double; };
264#endif
265
266#if defined(__FLT32_DIG__) && defined(_GLIBCXX_FLOAT_IS_IEEE_BINARY32)
267 template <>
268 struct __canonical_vec_type<_Float32>
269 { using type = float; };
270#endif
271
272 /** @internal
273 * @brief This ABI tag determines the data member(s) of basic_vec and basic_mask.
274 *
275 * `_Nreg` determines the number of recursive basic_vec/basic_mask data members where `_Nreg` is
276 * equal to 1. With `_Nreg` equal to 1, the basic_vec/basic_mask holds one vector builtin ( `_Np`
277 * greater than 1) or a scalar (`_Np` equal to 1).
278 * @f$\lceil\frac{\mathtt{Np}}{\mathtt{Nreg}}\rceil@f$ therefore determines the number of elements
279 * in a register (except for a remainder where it can be smaller). If `_Np` equals `_Nreg`, (the
280 * aforementioned quotient is 1), then basic_vec (recursively) holds non-vector data members and
281 * basic_mask holds bools.
282 *
283 * The `_Var` parameter determines details about the data member in the one register case. Masks
284 * can be represented as vector masks (the default comparison result of GNU vector builtins),
285 * bit-masks as used by AVX-512, bit-masks as used by ARM SVE (not yet implemented), or a single
286 * bool (for the `_Np` equals 1 case). For basic_mask it determines the actual data layout and
287 * for basic_mask it determines the result of compares.
288 *
289 * @tparam _Np The number of elements.
290 * @tparam _Nreg The number of registers needed to store `_Np` elements.
291 * @tparam _Var Determines how complex value-types are laid out and whether mask types use
292 * bit-masks or vector-masks.
293 */
294 template <int _Np, int _Nreg, underlying_type_t<_AbiVariant> _Var>
295 struct _Abi
296 {
297 static constexpr int _S_size = _Np;
298
299 /** @internal
300 * The number of registers needed to represent one basic_vec for the element type that was
301 * used on ABI deduction.
302 *
303 * For _CxCtgus the value applies twice, once per reals and once per imags.
304 *
305 * Examples:
306 * - '_Abi< 8, 2>' for 'int' is 2x 128-bit
307 * - '_Abi< 9, 3>' for 'int' is 2x 128-bit and 1x 32-bit
308 * - '_Abi<10, 3>' for 'int' is 2x 128-bit and 1x 64-bit
309 * - '_Abi<10, 1>' for 'int' is 1x 512-bit
310 * - '_Abi<10, 2>' for 'int' is 1x 256-bit and 1x 64-bit
311 * - '_Abi< 8, 2, _CxIleav>' for 'complex<float>' is 2x 256-bit
312 * - '_Abi< 9, 2, _CxIleav>' for 'complex<float>' is 1x 512-bit and 1x 64-bit
313 * - '_Abi< 8, 1, _CxCtgus>' for 'complex<float>' is 2x 256-bit
314 */
315 static constexpr int _S_nreg = _Nreg;
316
317 static_assert(_S_size > 0);
318 static_assert(_S_nreg > 0);
319
320 static constexpr _AbiVariant _S_variant = static_cast<_AbiVariant>(_Var);
321
322 static constexpr bool _S_is_cx_ileav
323 = __filter_abi_variant(_S_variant, _AbiVariant::_CxIleav) == _AbiVariant::_CxIleav;
324
325 static constexpr bool _S_is_cx_ctgus
326 = __filter_abi_variant(_S_variant, _AbiVariant::_CxCtgus) == _AbiVariant::_CxCtgus;
327
328 static_assert(!(_S_is_cx_ileav && _S_is_cx_ctgus)); // can't be both
329
330 static_assert(_S_size >= _S_nreg || (_S_is_cx_ileav && _S_size * 2 >= _S_nreg));
331
332 static constexpr bool _S_is_bitmask
333 = __filter_abi_variant(_S_variant, _AbiVariant::_BitMask) == _AbiVariant::_BitMask;
334
335 static constexpr bool _S_is_vecmask = !_S_is_bitmask;
336
337 template <typename _Tp>
338 using _DataType = decltype([] {
339 static_assert(_S_nreg == 1);
340 if constexpr (_S_size == 1)
341 return __canonical_vec_type_t<_Tp>();
342 else
343 {
344 constexpr int __n = __bit_ceil(unsigned(_S_size));
345 using _Vp [[__gnu__::__vector_size__(sizeof(_Tp) * __n)]]
346 = __canonical_vec_type_t<_Tp>;
347 return _Vp();
348 }
349 }());
350
351 template <size_t _Bytes>
352 using _MaskDataType
353 = decltype([] {
354 static_assert(_S_nreg == 1);
355 if constexpr (_S_size == 1)
356 return bool();
357 else if constexpr (_S_is_vecmask)
358 {
359 constexpr unsigned __vbytes = _Bytes * __bit_ceil(unsigned(_S_size));
360 using _Vp [[__gnu__::__vector_size__(__vbytes)]] = __integer_from<_Bytes>;
361 return _Vp();
362 }
363 else if constexpr (_Nreg > 1)
364 return _InvalidInteger();
365 else
366 return _Bitmask<_S_size>();
367 }());
368
369 template <int _N2, int _Nreg2 = __div_ceil(_N2, _S_size)>
370 static consteval auto
371 _S_resize()
372 {
373 if constexpr (_N2 == 1)
374 return _Abi<1, 1, _Var>();
375 else
376 return _Abi<_N2, _Nreg2, _Var>();
377 }
378 };
379
380 /** @internal
381 * Alias for an _Abi specialization where the _AbiVariant bits are combined into a single integer
382 * value.
383 *
384 * Rationale: Consider diagnostic output and mangling of e.g. vec<int, 4> with AVX512. That's an
385 * alias for std::simd::basic_vec<int, std::simd::_Abi<4, 1, 1ull>>. If _AbiVariant were the
386 * template argument type of _Abi, the diagnostic output would be 'std::simd::basic_vec<int,
387 * std::simd::_Abi<4, 1, (std::simd::_AbiVariant)std::simd::_AbiVariant::_BitMask>>'. That's a lot
388 * longer, requires longer mangled names, and bakes the names of the enumerators into the ABI. As
389 * soon as bits of multiple _AbiVariants are combined, this becomes hard to parse for humans
390 * anyway.
391 */
392 template <int _Np, int _Nreg, _AbiVariant... _Vs>
393 using _Abi_t = _Abi<_Np, _Nreg, (static_cast<underlying_type_t<_AbiVariant>>(_Vs) | ... | 0)>;
394
395 /** @internal
396 * This type is used whenever ABI tag deduction can't give a useful answer.
397 */
398 struct _InvalidAbi
399 { static constexpr int _S_size = 0; };
400
401 /** @internal
402 * Satisfied if @p _Tp is a valid simd ABI tag. This is a necessary but not sufficient condition
403 * for an enabled basic_vec/basic_mask specialization.
404 */
405 template <typename _Tp>
406 concept __abi_tag
407 = same_as<decltype(_Tp::_S_variant), const _AbiVariant>
408 && (_Tp::_S_size >= _Tp::_S_nreg) && (_Tp::_S_nreg >= 1)
409 && requires(_Tp __x) {
410 { __x.template _S_resize<_Tp::_S_size, _Tp::_S_nreg>() } -> same_as<_Tp>;
411 };
412
413 /** @internal
414 * Satisfied if `_Tp` is a valid simd ABI tag and one element is stored per register (number of
415 * registers equals size).
416 */
417 template <typename _Tp>
418 concept __scalar_abi_tag
419 = same_as<_Tp, _Abi_t<_Tp::_S_size, _Tp::_S_size, _Tp::_S_variant>> && __abi_tag<_Tp>;
420
421 // Determine if math functions must *raise* floating-point exceptions.
422 // math_errhandling may expand to an extern symbol, in which case we must assume fp exceptions
423 // need to be considered. A conforming C library must define math_errhandling, but in case it
424 // isn't defined we simply use the fallback.
425 // The macro is used as default template argument (rather than in a requires-clause) so that a
426 // non-constant math_errhandling is a substitution failure in the immediate context instead of a
427 // hard error at the point of definition.
428#ifdef math_errhandling
429 template <int _ErrHandling = math_errhandling>
430 consteval bool
431 __handle_fpexcept_impl(int)
432 { return 0 != (_ErrHandling & MATH_ERREXCEPT); }
433#endif
434
435 // Fallback if math_errhandling doesn't work: implement correct exception behavior.
436 consteval bool
437 __handle_fpexcept_impl(float)
438 { return true; }
439
440 /** @internal
441 * This type can be used as a template parameter for avoiding ODR violations, where code needs to
442 * differ depending on optimization flags (mostly fp-math related).
443 */
444 struct _OptTraits
445 {
446 consteval bool
447 _M_test(int __bit) const
448 { return ((_M_build_flags >> __bit) & 1) == 1; }
449
450 // true iff floating-point operations can signal an exception (allow non-default handler)
451 consteval bool
452 _M_fp_may_signal() const
453 { return _M_test(0); }
454
455 // true iff floating-point operations can raise an exception flag
456 consteval bool
457 _M_fp_may_raise() const
458 { return _M_test(12); }
459
460 consteval bool
461 _M_fast_math() const
462 { return _M_test(1); }
463
464 consteval bool
465 _M_finite_math_only() const
466 { return _M_test(2); }
467
468 consteval bool
469 _M_no_signed_zeros() const
470 { return _M_test(3); }
471
472 consteval bool
473 _M_signed_zeros() const
474 { return !_M_test(3); }
475
476 consteval bool
477 _M_reciprocal_math() const
478 { return _M_test(4); }
479
480 consteval bool
481 _M_no_math_errno() const
482 { return _M_test(5); }
483
484 consteval bool
485 _M_math_errno() const
486 { return !_M_test(5); }
487
488 consteval bool
489 _M_associative_math() const
490 { return _M_test(6); }
491
492 consteval bool
493 _M_conforming_to_STDC_annex_G() const
494 { return _M_test(10) && !_M_finite_math_only(); }
495
496 consteval bool
497 _M_support_snan() const
498 { return _M_test(11); }
499
500 __UINT64_TYPE__ _M_build_flags
501 = 0
502#if !__NO_TRAPPING_MATH__
503 + (1 << 0)
504#endif
505 + (__handle_fpexcept_impl(0) << 12)
506#if __FAST_MATH__
507 + (1 << 1)
508#endif
509#if __FINITE_MATH_ONLY__
510 + (1 << 2)
511#endif
512#if __NO_SIGNED_ZEROS__
513 + (1 << 3)
514#endif
515#if __RECIPROCAL_MATH__
516 + (1 << 4)
517#endif
518#if __NO_MATH_ERRNO__
519 + (1 << 5)
520#endif
521#if __ASSOCIATIVE_MATH__
522 + (1 << 6)
523#endif
524 // bits 7, 8, and 9 reserved for __FLT_EVAL_METHOD__
525#if __FLT_EVAL_METHOD__ == 1
526 + (1 << 7)
527#elif __FLT_EVAL_METHOD__ == 2
528 + (2 << 7)
529#elif __FLT_EVAL_METHOD__ != 0
530 + (3 << 7)
531#endif
532
533 // C Annex G defines the behavior of complex<T> where T is IEC60559 floating-point. If
534 // __STDC_IEC_60559_COMPLEX__ is defined then Annex G is implemented - and simd<complex>
535 // will do so as well. However, Clang never defines the macro.
536#if defined __STDC_IEC_60559_COMPLEX__ || defined __STDC_IEC_559_COMPLEX__ || defined _GLIBCXX_CLANG
537 + (1 << 10)
538#endif
539#if __SUPPORT_SNAN__
540 + (1 << 11)
541#endif
542 ;
543 };
544
545 /** @internal
546 * Return true iff @p __s equals "1".
547 */
548 consteval bool
549 __streq_to_1(const char* __s)
550 { return __s != nullptr && __s[0] == '1' && __s[1] == '\0'; }
551
552 /** @internal
553 * If the macro given as @p feat is defined to 1, expands to a bit set at position @p off.
554 * Otherwise, expand to zero.
555 */
556#define _GLIBCXX_SIMD_ARCH_FLAG(off, feat) \
557 (static_cast<__UINT64_TYPE__>(std::simd::__streq_to_1(_GLIBCXX_SIMD_TOSTRING_IMPL(feat))) << off)
558
559#if _GLIBCXX_X86
560
561#define _GLIBCXX_SIMD_ARCH_TRAITS_INIT { \
562 _GLIBCXX_SIMD_ARCH_FLAG(0, __MMX__) \
563 | _GLIBCXX_SIMD_ARCH_FLAG( 1, __SSE__) \
564 | _GLIBCXX_SIMD_ARCH_FLAG( 2, __SSE2__) \
565 | _GLIBCXX_SIMD_ARCH_FLAG( 3, __SSE3__) \
566 | _GLIBCXX_SIMD_ARCH_FLAG( 4, __SSSE3__) \
567 | _GLIBCXX_SIMD_ARCH_FLAG( 5, __SSE4_1__) \
568 | _GLIBCXX_SIMD_ARCH_FLAG( 6, __SSE4_2__) \
569 | _GLIBCXX_SIMD_ARCH_FLAG( 7, __POPCNT__) \
570 | _GLIBCXX_SIMD_ARCH_FLAG( 8, __AVX__) \
571 | _GLIBCXX_SIMD_ARCH_FLAG( 9, __F16C__) \
572 | _GLIBCXX_SIMD_ARCH_FLAG(10, __BMI__) \
573 | _GLIBCXX_SIMD_ARCH_FLAG(11, __BMI2__) \
574 | _GLIBCXX_SIMD_ARCH_FLAG(12, __LZCNT__) \
575 | _GLIBCXX_SIMD_ARCH_FLAG(13, __AVX2__) \
576 | _GLIBCXX_SIMD_ARCH_FLAG(14, __FMA__) \
577 | _GLIBCXX_SIMD_ARCH_FLAG(15, __AVX512F__) \
578 | _GLIBCXX_SIMD_ARCH_FLAG(16, __AVX512CD__) \
579 | _GLIBCXX_SIMD_ARCH_FLAG(17, __AVX512DQ__) \
580 | _GLIBCXX_SIMD_ARCH_FLAG(18, __AVX512BW__) \
581 | _GLIBCXX_SIMD_ARCH_FLAG(19, __AVX512VL__) \
582 | _GLIBCXX_SIMD_ARCH_FLAG(20, __AVX512BITALG__) \
583 | _GLIBCXX_SIMD_ARCH_FLAG(21, __AVX512VBMI__) \
584 | _GLIBCXX_SIMD_ARCH_FLAG(22, __AVX512VBMI2__) \
585 | _GLIBCXX_SIMD_ARCH_FLAG(23, __AVX512IFMA__) \
586 | _GLIBCXX_SIMD_ARCH_FLAG(24, __AVX512VNNI__) \
587 | _GLIBCXX_SIMD_ARCH_FLAG(25, __AVX512VPOPCNTDQ__) \
588 | _GLIBCXX_SIMD_ARCH_FLAG(26, __AVX512FP16__) \
589 | _GLIBCXX_SIMD_ARCH_FLAG(27, __AVX512BF16__) \
590 | _GLIBCXX_SIMD_ARCH_FLAG(28, __AVXIFMA__) \
591 | _GLIBCXX_SIMD_ARCH_FLAG(29, __AVXNECONVERT__) \
592 | _GLIBCXX_SIMD_ARCH_FLAG(30, __AVXVNNI__) \
593 | _GLIBCXX_SIMD_ARCH_FLAG(31, __AVXVNNIINT8__) \
594 | _GLIBCXX_SIMD_ARCH_FLAG(32, __AVXVNNIINT16__) \
595 | _GLIBCXX_SIMD_ARCH_FLAG(33, __AVX10_1__) \
596 | _GLIBCXX_SIMD_ARCH_FLAG(34, __AVX10_2__) \
597 | _GLIBCXX_SIMD_ARCH_FLAG(35, __AVX512VP2INTERSECT__) \
598 | _GLIBCXX_SIMD_ARCH_FLAG(36, __SSE4A__) \
599 | _GLIBCXX_SIMD_ARCH_FLAG(37, __FMA4__) \
600 | _GLIBCXX_SIMD_ARCH_FLAG(38, __XOP__) \
601 }
602 // Should this include __APX_F__? I don't think it's relevant for use in constexpr-if branches =>
603 // no ODR issue? The same could be said about several other flags above that are not checked
604 // anywhere.
605
606 struct _ArchTraits
607 {
608 __UINT64_TYPE__ _M_flags = _GLIBCXX_SIMD_ARCH_TRAITS_INIT;
609
610 consteval bool
611 _M_test(int __bit) const
612 { return ((_M_flags >> __bit) & 1) == 1; }
613
614 consteval bool
615 _M_have_mmx() const
616 { return _M_test(0); }
617
618 consteval bool
619 _M_have_sse() const
620 { return _M_test(1); }
621
622 consteval bool
623 _M_have_sse2() const
624 { return _M_test(2); }
625
626 consteval bool
627 _M_have_sse3() const
628 { return _M_test(3); }
629
630 consteval bool
631 _M_have_ssse3() const
632 { return _M_test(4); }
633
634 consteval bool
635 _M_have_sse4_1() const
636 { return _M_test(5); }
637
638 consteval bool
639 _M_have_sse4_2() const
640 { return _M_test(6); }
641
642 consteval bool
643 _M_have_popcnt() const
644 { return _M_test(7); }
645
646 consteval bool
647 _M_have_avx() const
648 { return _M_test(8); }
649
650 consteval bool
651 _M_have_f16c() const
652 { return _M_test(9); }
653
654 consteval bool
655 _M_have_bmi() const
656 { return _M_test(10); }
657
658 consteval bool
659 _M_have_bmi2() const
660 { return _M_test(11); }
661
662 consteval bool
663 _M_have_lzcnt() const
664 { return _M_test(12); }
665
666 consteval bool
667 _M_have_avx2() const
668 { return _M_test(13); }
669
670 consteval bool
671 _M_have_fma() const
672 { return _M_test(14); }
673
674 consteval bool
675 _M_have_avx512f() const
676 { return _M_test(15); }
677
678 consteval bool
679 _M_have_avx512cd() const
680 { return _M_test(16); }
681
682 consteval bool
683 _M_have_avx512dq() const
684 { return _M_test(17); }
685
686 consteval bool
687 _M_have_avx512bw() const
688 { return _M_test(18); }
689
690 consteval bool
691 _M_have_avx512vl() const
692 { return _M_test(19); }
693
694 consteval bool
695 _M_have_avx512bitalg() const
696 { return _M_test(20); }
697
698 consteval bool
699 _M_have_avx512vbmi() const
700 { return _M_test(21); }
701
702 consteval bool
703 _M_have_avx512vbmi2() const
704 { return _M_test(22); }
705
706 consteval bool
707 _M_have_avx512ifma() const
708 { return _M_test(23); }
709
710 consteval bool
711 _M_have_avx512vnni() const
712 { return _M_test(24); }
713
714 consteval bool
715 _M_have_avx512vpopcntdq() const
716 { return _M_test(25); }
717
718 consteval bool
719 _M_have_avx512fp16() const
720 { return _M_test(26); }
721
722 consteval bool
723 _M_have_avx512bf16() const
724 { return _M_test(27); }
725
726 consteval bool
727 _M_have_avxifma() const
728 { return _M_test(28); }
729
730 consteval bool
731 _M_have_avxneconvert() const
732 { return _M_test(29); }
733
734 consteval bool
735 _M_have_avxvnni() const
736 { return _M_test(30); }
737
738 consteval bool
739 _M_have_avxvnniint8() const
740 { return _M_test(31); }
741
742 consteval bool
743 _M_have_avxvnniint16() const
744 { return _M_test(32); }
745
746 consteval bool
747 _M_have_avx10_1() const
748 { return _M_test(33); }
749
750 consteval bool
751 _M_have_avx10_2() const
752 { return _M_test(34); }
753
754 consteval bool
755 _M_have_avx512vp2intersect() const
756 { return _M_test(35); }
757
758 consteval bool
759 _M_have_sse4a() const
760 { return _M_test(36); }
761
762 consteval bool
763 _M_have_fma4() const
764 { return _M_test(37); }
765
766 consteval bool
767 _M_have_xop() const
768 { return _M_test(38); }
769
770 template <typename _Tp>
771 consteval bool
772 _M_eval_as_f32() const
773 { return is_same_v<_Tp, _Float16> && !_M_have_avx512fp16(); }
774
775 consteval bool
776 _M_have_addsub() const
777 { return _M_have_sse3(); }
778 };
779
780 template <typename _Tp, _ArchTraits _Traits = {}>
781 consteval auto
782 __native_abi()
783 {
784 constexpr int __adj_sizeof = sizeof(_Tp) * (1 + is_same_v<_Tp, _Float16>);
785 if constexpr (!__vectorizable<_Tp>)
786 return _InvalidAbi();
787 else if constexpr (__complex_like<_Tp>)
788 {
789 constexpr auto __underlying = std::simd::__native_abi<typename _Tp::value_type>();
790 constexpr int __cx_size = __underlying._S_size / (__underlying._S_size == 1 ? 1 : 2);
791 return _Abi_t<__cx_size, 1, __underlying._S_variant, _AbiVariant::_CxIleav>();
792 }
793 else if constexpr (_Traits._M_have_avx512fp16())
794 return _Abi_t<64 / sizeof(_Tp), 1, _AbiVariant::_BitMask>();
795 else if constexpr (_Traits._M_have_avx512f())
796 return _Abi_t<64 / __adj_sizeof, 1, _AbiVariant::_BitMask>();
797 else if constexpr (is_same_v<_Tp, _Float16> && !_Traits._M_have_f16c())
798 return _Abi_t<1, 1>();
799 else if constexpr (_Traits._M_have_avx2())
800 return _Abi_t<32 / __adj_sizeof, 1>();
801 else if constexpr (_Traits._M_have_avx() && is_floating_point_v<_Tp>)
802 return _Abi_t<32 / __adj_sizeof, 1>();
803 else if constexpr (_Traits._M_have_sse2())
804 return _Abi_t<16 / __adj_sizeof, 1>();
805 else if constexpr (_Traits._M_have_sse() && is_floating_point_v<_Tp>
806 && sizeof(_Tp) == sizeof(float))
807 return _Abi_t<16 / __adj_sizeof, 1>();
808 // no MMX: we can't emit EMMS where it would be necessary
809 else
810 return _Abi_t<1, 1>();
811 }
812
813#else
814
815 // scalar fallback
816 struct _ArchTraits
817 {
818 __UINT64_TYPE__ _M_flags = 0;
819
820 constexpr bool
821 _M_test(int __bit) const
822 { return ((_M_flags >> __bit) & 1) == 1; }
823 };
824
825 template <typename _Tp>
826 consteval auto
827 __native_abi()
828 {
829 if constexpr (!__vectorizable<_Tp>)
830 return _InvalidAbi();
831 else if constexpr (__complex_like<_Tp>)
832 return _Abi_t<1, 1, _AbiVariant::_CxIleav>();
833 else
834 return _Abi_t<1, 1>();
835 }
836
837#endif
838
839 /** @internal
840 * You must use this type as template argument to function templates that are not declared
841 * always_inline (to avoid issues when linking code compiled with different compiler flags).
842 */
843 struct _TargetTraits
844 : _ArchTraits, _OptTraits
845 {};
846
847 /** @internal
848 * Alias for an ABI tag such that basic_vec<_Tp, __native_abi_t_<_Tp>> stores one SIMD register of
849 * optimal width.
850 *
851 * @tparam _Tp A vectorizable type.
852 *
853 * C++26 [simd.expos.abi]
854 */
855 template <typename _Tp>
856 using __native_abi_t = decltype(std::simd::__native_abi<_Tp>());
857
858 template <typename _Tp, int _Np, _TargetTraits _Target = {}>
859 consteval auto
860 __deduce_abi()
861 {
862 constexpr auto __native = std::simd::__native_abi<_Tp>();
863 if constexpr (0 == __native._S_size || _Np <= 0)
864 return _InvalidAbi();
865 else if constexpr (_Np == __native._S_size)
866 return __native;
867 else
868 return __native.template _S_resize<_Np>();
869 }
870
871 /** @internal
872 * Alias for an ABI tag @c A such that `basic_vec<_Tp, A>` stores @p _Np elements.
873 *
874 * C++26 [simd.expos.abi]
875 */
876 template <typename _Tp, int _Np>
877 using __deduce_abi_t = decltype(std::simd::__deduce_abi<_Tp, _Np>());
878
879 /** @internal
880 * \c rebind implementation detail for basic_vec, and basic_mask where we know the destination
881 * value-type
882 */
883 template <typename _Tp, int _Np, __abi_tag _A0, _ArchTraits = {}>
884 consteval auto
885 __abi_rebind()
886 {
887 if constexpr (_Np <= 0 || !__vectorizable<_Tp>)
888 return _InvalidAbi();
889
890 else
891 {
892 using _Native = remove_const_t<decltype(std::simd::__native_abi<_Tp>())>;
893 static_assert(0 != _Native::_S_size);
894 constexpr int __nreg = __div_ceil(_Np, _Native::_S_size);
895
896 // __scalar_abi_tag is sticky (unless we reach size 1, where we can't know whether it was
897 // an explicit __scalar_abi_tag before some resize_t)
898 if constexpr (__scalar_abi_tag<_Native> || (__scalar_abi_tag<_A0> && _A0::_S_size >= 2))
899 {
900 constexpr bool __remove_cx
901 = __filter_abi_variant(_A0::_S_variant, _AbiVariant::_CxVariants) != _AbiVariant()
902 && !__complex_like<_Tp>;
903 constexpr bool __add_cx
904 = __filter_abi_variant(_A0::_S_variant, _AbiVariant::_CxVariants) == _AbiVariant()
905 && __complex_like<_Tp>;
906
907 if constexpr (__remove_cx)
908 return _Abi_t<_Np, _Np,
909 __filter_abi_variant(_A0::_S_variant, _AbiVariant::_MaskVariants)>();
910 else if constexpr (__add_cx)
911 return _Abi_t<_Np, _Np, _A0::_S_variant,
912 __filter_abi_variant(_Native::_S_variant, _AbiVariant::_CxVariants)>();
913 else
914 return _A0::template _S_resize<_Np, _Np>();
915 }
916
917 else if constexpr (__complex_like<_Tp> && _A0::_S_is_cx_ctgus && _Native::_S_is_cx_ileav)
918 // we need half the number of registers since the number applies twice, to reals and
919 // imaginaries.
920 return _A0::template _S_resize<_Np, __div_ceil(__nreg, 2)>();
921
922 else if constexpr (__complex_like<_Tp> && _A0::_S_is_cx_ileav && _Native::_S_is_cx_ctgus)
923 return _A0::template _S_resize<_Np, __nreg * 2>();
924
925 else if constexpr (__complex_like<_Tp> && (_A0::_S_is_cx_ctgus || _A0::_S_is_cx_ileav))
926 return _A0::template _S_resize<_Np, __nreg>();
927
928 else if constexpr (__complex_like<_Tp>)
929 // Bit vs. Vec Mask determined by _A0, _CxVariant determined by _Native
930 return _Abi_t<_Native::_S_size, 1, _A0::_S_variant,
931 __filter_abi_variant(_Native::_S_variant, _AbiVariant::_CxVariants)>
932 ::template _S_resize<_Np, __nreg>();
933
934 else
935 return _Abi_t<_Native::_S_size, 1, __filter_abi_variant(_A0::_S_variant,
936 _AbiVariant::_MaskVariants)
937 >::template _S_resize<_Np, __nreg>();
938 }
939 }
940
941 /** @internal
942 * @c rebind implementation detail for basic_mask.
943 *
944 * The important difference here is that we have no information about the actual value-type other
945 * than its @c sizeof. So `_Bytes == 8` could mean `complex<float>`, @c double, or @c int64_t.
946 * E.g. `_Np == 4` with AVX w/o AVX2 that's `vector(4) int`, `vector(4) long long`, or `2x
947 * vector(2) long long`.
948 * That's why this overload has the additional @p _IsOnlyResize parameter, which tells us that the
949 * value-type doesn't change.
950 */
951 template <size_t _Bytes, int _Np, __abi_tag _A0, bool _IsOnlyResize, _ArchTraits _Traits = {}>
952 consteval auto
953 __abi_rebind()
954 {
955 constexpr bool __from_cx = _A0::_S_is_cx_ctgus || _A0::_S_is_cx_ileav;
956
957 if constexpr (_Bytes == 0 || _Np <= 0)
958 return _InvalidAbi();
959
960 // If the source ABI is complex, _Bytes == sizeof(complex<float>) or
961 // sizeof(complex<float16_t>), and _IsOnlyResize is true, then it's a mask<complex<float>,
962 // _Np>
963 else if constexpr (__from_cx && _IsOnlyResize && _Bytes == 2 * sizeof(double))
964 return __abi_rebind<complex<double>, _Np, _A0>();
965 else if constexpr (__from_cx && _IsOnlyResize && _Bytes == 2 * sizeof(float))
966 return __abi_rebind<complex<float>, _Np, _A0>();
967 else if constexpr (__from_cx && _IsOnlyResize && _Bytes == 2 * sizeof(_Float16))
968 return __abi_rebind<complex<_Float16>, _Np, _A0>();
969
970#if _GLIBCXX_X86
971 // AVX w/o AVX2:
972 // e.g. resize_t<8, mask<float, Whatever>> needs to be _Abi<8, 1> not _Abi<8, 2>
973 // We determine whether _A0 identifies an AVX vector by looking at the size of a native
974 // register. If it's 32, it's a YMM register, otherwise it's 16 or less.
975 else if constexpr (_IsOnlyResize
976 && _Traits._M_have_avx() && !_Traits._M_have_avx2()
977 && __bit_ceil(__div_ceil<unsigned>(
978 _A0::_S_size, _A0::_S_nreg)) * _Bytes == 32)
979 {
980 if constexpr (_Bytes == sizeof(double))
981 return __abi_rebind<double, _Np, _A0>();
982 else if constexpr (_Bytes == sizeof(float))
983 return __abi_rebind<float, _Np, _A0>();
984 else if constexpr (_Traits._M_have_f16c() && _Bytes == sizeof(_Float16))
985 return __abi_rebind<_Float16, _Np, _A0>();
986 else // impossible
987 static_assert(false);
988 }
989#endif
990
991 else
992 return __abi_rebind<__integer_from<_Bytes>, _Np, _A0>();
993 }
994
995 /** @internal
996 * Returns true unless _GLIBCXX_SIMD_COND_EXPLICIT_MASK_CONVERSION is defined.
997 *
998 * On IvyBridge, (vec<float> == 0.f) == (rebind_t<int, vec<float>> == 0) does not compile. It does
999 * compile on basically every other target, though. This is due to the difference in ABI tag:
1000 * _Abi<8, 1, [...]> vs. _Abi<8, 2, [...]> (8 elements, 1 vs. 2 registers).
1001 * I know how to define this function for libstdc++ to avoid interconvertible masks. The question
1002 * is whether we can specify this in general for C++29.
1003 *
1004 * Idea: Is rebind_t<integer-from<...>, mask>::abi_type the same type as
1005 * deduce-t<integer-from<...>, mask::size()>? If yes, it's the "better" ABI tag. However, this
1006 * makes the conversion behavior dependent on compiler flags. Probably not what we want.
1007 */
1008 template <typename _To, typename _From>
1009 consteval bool
1010 __is_mask_conversion_explicit([[maybe_unused]] size_t __b0, [[maybe_unused]] size_t __b1)
1011 {
1012 constexpr int __n = _To::_S_size;
1013 static_assert(__n == _From::_S_size);
1014#ifndef _GLIBCXX_SIMD_COND_EXPLICIT_MASK_CONVERSION
1015 /// C++26 [simd.mask.ctor] uses unconditional explicit
1016 return true;
1017#else
1018 if (__b0 != __b1)
1019 return true;
1020
1021 // converting to a bit-mask is better
1022 else if constexpr (_To::_S_is_vecmask != _From::_S_is_vecmask)
1023 return _To::_S_is_vecmask; // to vector-mask is explicit
1024
1025 // with vec-masks, fewer registers is better
1026 else if constexpr (_From::_S_nreg != _To::_S_nreg)
1027 return _From::_S_nreg < _To::_S_nreg;
1028
1029 // differ only on _Cx flags
1030 // interleaved complex is worse
1031 else if constexpr (_To::_S_is_cx_ileav)
1032 return true;
1033 else if constexpr (_From::_S_is_cx_ileav)
1034 return false;
1035
1036 // prefer non-_Cx over _CxCtgus
1037 else if constexpr (_To::_S_is_cx_ctgus)
1038 return true;
1039 else if constexpr (_From::_S_is_cx_ctgus)
1040 return false;
1041 else
1042 __builtin_unreachable();
1043#endif
1044 }
1045
1046 /** @internal
1047 * An alias for a signed integer type.
1048 *
1049 * libstdc++ unconditionally uses @c int here, since it matches the return type of
1050 * 'Bit Operation Builtins' in GCC.
1051 *
1052 * C++26 [simd.expos.defn]
1053 */
1054 using __simd_size_type = int;
1055
1056 // integral_constant shortcut
1057 template <__simd_size_type _Xp>
1058 inline constexpr integral_constant<__simd_size_type, _Xp> __simd_size_c = {};
1059
1060 // [simd.syn]
1061 template <typename _Tp, typename _Ap = __native_abi_t<_Tp>>
1062 class basic_vec;
1063
1064 template <typename _Tp, __simd_size_type _Np = __native_abi_t<_Tp>::_S_size>
1065 using vec = basic_vec<_Tp, __deduce_abi_t<_Tp, _Np>>;
1066
1067 template <size_t _Bytes, typename _Ap = __native_abi_t<__integer_from<_Bytes>>>
1068 class basic_mask;
1069
1070 template <typename _Tp, __simd_size_type _Np = __native_abi_t<_Tp>::_S_size>
1071 using mask = basic_mask<sizeof(_Tp), __deduce_abi_t<_Tp, _Np>>;
1072
1073 // [simd.ctor] load constructor constraints
1074 template <typename _Rg>
1075 consteval size_t
1076 __static_range_size(_Rg& __r)
1077 {
1078 if constexpr (ranges::__static_sized_range<_Rg>)
1079 return ranges::size(__r);
1080 else
1081 return dynamic_extent;
1082 }
1083
1084 // [simd.general] value-preserving
1085 template <typename _From, typename _To>
1086 concept __arithmetic_only_value_preserving_convertible_to
1087 = convertible_to<_From, _To> && is_arithmetic_v<_From> && is_arithmetic_v<_To>
1088 && !(is_signed_v<_From> && is_unsigned_v<_To>)
1092
1093 /** @internal
1094 * Satisfied if the conversion from @p _From to @p _To is a value-preserving conversion.
1095 *
1096 * C++26 [simd.general]
1097 */
1098 template <typename _From, typename _To>
1099 concept __value_preserving_convertible_to
1100 = __arithmetic_only_value_preserving_convertible_to<_From, _To>
1101 || (__complex_like<_To> && __arithmetic_only_value_preserving_convertible_to<
1102 _From, typename _To::value_type>);
1103
1104 // LWG4420
1105 template <typename _From, typename _To>
1106 concept __explicitly_convertible_to = requires {
1107 static_cast<_To>(declval<_From>());
1108 };
1109
1110 /** @internal
1111 * C++26 [simd.expos]
1112 */
1113 // [simd.ctor] explicit(...) of broadcast ctor
1114 template <auto _From, typename _To>
1115 concept __non_narrowing_constexpr_conversion
1116 = is_arithmetic_v<decltype(_From)>
1117 && static_cast<decltype(_From)>(static_cast<_To>(_From)) == _From
1118 && !(unsigned_integral<_To> && _From < decltype(_From)())
1119 && _From <= std::numeric_limits<_To>::max()
1121
1122 // [simd.ctor] p4
1123 // This implements LWG4436 (submitted on 2025-10-28)
1124 template <typename _From, typename _To>
1125 concept __broadcast_constructible
1126 = ((convertible_to<_From, _To> && !is_arithmetic_v<remove_cvref_t<_From>>
1127 && !__constexpr_wrapper_like<remove_cvref_t<_From>>) // 4.1
1128 || __value_preserving_convertible_to<remove_cvref_t<_From>, _To> // 4.2
1129 || (__constexpr_wrapper_like<remove_cvref_t<_From>> // 4.3
1130 && __non_narrowing_constexpr_conversion<auto(remove_cvref_t<_From>::value),
1131 _To>));
1132
1133 // __higher_floating_point_rank_than<_Tp, U> (_Tp has higher or equal floating point rank than U)
1134 template <typename _From, typename _To>
1135 consteval bool
1136 __higher_floating_point_rank_than()
1137 {
1138 return floating_point<_From> && floating_point<_To>
1139 && is_same_v<common_type_t<_From, _To>, _From> && !is_same_v<_From, _To>;
1140 }
1141
1142 // __higher_integer_rank_than<_Tp, U> (_Tp has higher or equal integer rank than U)
1143 template <typename _From, typename _To>
1144 consteval bool
1145 __higher_integer_rank_than()
1146 {
1147 return integral<_From> && integral<_To>
1148 && (sizeof(_From) > sizeof(_To) || is_same_v<common_type_t<_From, _To>, _From>)
1149 && !is_same_v<_From, _To>;
1150 }
1151
1152 template <typename _From, typename _To>
1153 concept __higher_rank_than
1154 = __higher_floating_point_rank_than<_From, _To>() || __higher_integer_rank_than<_From, _To>();
1155
1156 struct __convert_flag;
1157
1158 template <typename _From, typename _To, typename... _Flags>
1159 concept __loadstore_convertible_to
1160 = same_as<_From, _To>
1161 || (__vectorizable<_From> && __vectorizable<_To>
1162 && (__value_preserving_convertible_to<_From, _To>
1163 || (__explicitly_convertible_to<_From, _To>
1164 && (std::is_same_v<_Flags, __convert_flag> || ...))));
1165
1166 template <typename _From, typename _To>
1167 concept __simd_generator_convertible_to
1168 = std::convertible_to<_From, _To>
1169 && (!is_arithmetic_v<_From> || __value_preserving_convertible_to<_From, _To>);
1170
1171 template <typename _Fp, typename _Tp, __simd_size_type... _Is>
1172 requires (__simd_generator_convertible_to<
1173 decltype(declval<_Fp>()(__simd_size_c<_Is>)), _Tp> && ...)
1174 constexpr void
1175 __simd_generator_invokable_impl(integer_sequence<__simd_size_type, _Is...>);
1176
1177 template <typename _Fp, typename _Tp, __simd_size_type _Np>
1178 concept __simd_generator_invokable = requires {
1179 __simd_generator_invokable_impl<_Fp, _Tp>(make_integer_sequence<__simd_size_type, _Np>());
1180 };
1181
1182 template <typename _Fp>
1183 concept __index_permutation_function_sized = requires(_Fp const& __f)
1184 {
1185 { __f(0, 0) } -> std::integral;
1186 };
1187
1188 template <typename _Fp, typename _Simd>
1189 concept __index_permutation_function
1190 = __index_permutation_function_sized<_Fp> || requires(_Fp const& __f) {
1191 { __f(0) } -> std::integral;
1192 };
1193
1194 /** @internal
1195 * The value of the @c _Bytes template argument to a @c basic_mask specialization.
1196 *
1197 * C++26 [simd.expos.defn]
1198 */
1199 template <typename _Tp>
1200 constexpr size_t __mask_element_size = 0;
1201
1202 template <size_t _Bytes, __abi_tag _Ap>
1203 constexpr size_t __mask_element_size<basic_mask<_Bytes, _Ap>> = _Bytes;
1204
1205 // [simd.expos]
1206 template <typename _Vp>
1207 concept __simd_vec_type
1208 = same_as<_Vp, basic_vec<typename _Vp::value_type, typename _Vp::abi_type>>
1209 && is_default_constructible_v<_Vp>;
1210
1211 template <typename _Vp>
1212 concept __simd_mask_type
1213 = same_as<_Vp, basic_mask<__mask_element_size<_Vp>, typename _Vp::abi_type>>
1214 && is_default_constructible_v<_Vp>;
1215
1216 /** @internal
1217 * Satisfied if @p _Tp is a data-parallel type.
1218 */
1219 template <typename _Vp>
1220 concept __simd_vec_or_mask_type = __simd_vec_type<_Vp> || __simd_mask_type<_Vp>;
1221
1222 template <typename _Vp>
1223 concept __simd_floating_point
1224 = __simd_vec_type<_Vp> && floating_point<typename _Vp::value_type>;
1225
1226 template <typename _Vp>
1227 concept __simd_integral
1228 = __simd_vec_type<_Vp> && integral<typename _Vp::value_type>;
1229
1230 template <typename _Vp>
1231 concept __simd_unsigned_integer
1232 = __simd_vec_type<_Vp> && __unsigned_integer<typename _Vp::value_type>;
1233
1234 template <typename _Vp>
1235 using __simd_complex_value_type = typename _Vp::value_type::value_type;
1236
1237 template <typename _Vp>
1238 concept __simd_complex
1239 = __simd_vec_type<_Vp> && __complex_like_impl<typename _Vp::value_type>;
1240
1241 template <typename _Tp>
1242 concept __converts_to_vec
1243 = __simd_vec_type<decltype(declval<const _Tp&>() + declval<const _Tp&>())>;
1244
1245 template <__converts_to_vec _Tp>
1246 using __deduced_vec_t = decltype(declval<const _Tp&>() + declval<const _Tp&>());
1247
1248 template <typename _Vp, typename _Tp>
1249 using __make_compatible_simd_t
1250 = decltype([] {
1251 using _Up = decltype(declval<const _Tp&>() + declval<const _Tp&>());
1252 if constexpr (__simd_vec_type<_Up>)
1253 return _Up();
1254 else
1255 return vec<_Up, _Vp::size()>();
1256 }());
1257
1258 template <typename _Tp>
1259 concept __math_floating_point = __simd_floating_point<__deduced_vec_t<_Tp>>;
1260
1261 template <typename _BinaryOperation, typename _Tp>
1262 concept __reduction_binary_operation
1263 = requires (const _BinaryOperation __binary_op, const vec<_Tp, 1> __v) {
1264 { __binary_op(__v, __v) } -> same_as<vec<_Tp, 1>>;
1265 };
1266
1267 /** @internal
1268 * Returns the highest index @c i where `(__bits >> i) & 1` equals @c 1.
1269 */
1270 [[__gnu__::__always_inline__]]
1271 constexpr __simd_size_type
1272 __highest_bit(std::unsigned_integral auto __bits)
1273 {
1275 constexpr auto _Nd = __int_traits<decltype(__bits)>::__digits;
1276 return _Nd - 1 - __countl_zero(__bits);
1277 }
1278
1279 template <__vectorizable _Tp, __simd_size_type _Np, __abi_tag _Ap>
1280 using __similar_mask = basic_mask<sizeof(_Tp), decltype(__abi_rebind<_Tp, _Np, _Ap>())>;
1281
1282 template <size_t _Bytes, __abi_tag _Ap>
1283 using __component_mask_for_ileav
1284 = basic_mask<_Bytes / 2,
1285 decltype(__abi_rebind<__float_from<_Bytes / 2>, _Ap::_S_size * 2, _Ap>())>;
1286
1287 template <size_t _Bytes, __abi_tag _Ap>
1288 using __component_mask_for_ctgus
1289 = basic_mask<_Bytes / 2,
1290 decltype(__abi_rebind<__float_from<_Bytes / 2>, _Ap::_S_size, _Ap>())>;
1291
1292 // Allow _Tp to be _InvalidInteger for __integer_from<16>
1293 template <typename _Tp, __simd_size_type _Np, __abi_tag _Ap>
1294 using __similar_vec = basic_vec<_Tp, decltype(__abi_rebind<_Tp, _Np, _Ap>())>;
1295
1296 // LWG4470 [simd.expos]
1297 template <size_t _Bytes, typename _Ap>
1298 using __simd_vec_from_mask_t = __similar_vec<__integer_from<_Bytes>, _Ap::_S_size, _Ap>;
1299
1300#if _GLIBCXX_SIMD_THROW_ON_BAD_VALUE // used for unit tests (also see P3844)
1301 class __bad_value_preserving_cast
1302 {};
1303
1304#define __glibcxx_on_bad_value_preserving_cast throw __bad_value_preserving_cast
1305#else
1306 void __bad_value_preserving_cast(); // not defined
1307
1308#define __glibcxx_on_bad_value_preserving_cast __bad_value_preserving_cast
1309#endif
1310
1311 template <typename _To, typename _From>
1312#if _GLIBCXX_SIMD_THROW_ON_BAD_VALUE // see P3844
1313 [[__gnu__::__optimize__("exceptions")]] // work around potential -fno-exceptions
1314#endif
1315 consteval _To
1316 __value_preserving_cast(const _From& __x)
1317 {
1318 static_assert(is_arithmetic_v<_From>);
1319 if constexpr (!__value_preserving_convertible_to<_From, _To>)
1320 {
1321 using _Up = typename __make_unsigned<_From>::__type;
1322 if (static_cast<_Up>(static_cast<_To>(__x)) != static_cast<_Up>(__x))
1323 __glibcxx_on_bad_value_preserving_cast();
1324 else if constexpr (is_signed_v<_From> && is_unsigned_v<_To>)
1325 {
1326 if (__x < _From())
1327 __glibcxx_on_bad_value_preserving_cast();
1328 }
1329 else if constexpr (unsigned_integral<_From> && signed_integral<_To>)
1330 {
1331 if (__x > numeric_limits<_To>::max())
1332 __glibcxx_on_bad_value_preserving_cast();
1333 }
1334 }
1335 return static_cast<_To>(__x);
1336 }
1337
1338 /** @internal
1339 * std::pair is not trivially copyable, this one is
1340 */
1341 template <typename _T0, typename _T1>
1342 struct __trivial_pair
1343 {
1344 _T0 _M_first;
1345 _T1 _M_second;
1346 };
1347
1348 template <typename _From, typename _To>
1349 concept __converts_trivially = convertible_to<_From, _To>
1350 && sizeof(_From) == sizeof(_To)
1351 && is_integral_v<_From> == is_integral_v<_To>
1352 && is_floating_point_v<_From> == is_floating_point_v<_To>;
1353
1354 [[__gnu__::__always_inline__]]
1355 constexpr void
1356 __bit_foreach(unsigned_integral auto __bits, auto&& __fun)
1357 {
1358 static_assert(sizeof(__bits) >= sizeof(int)); // avoid promotion to int
1359 while (__bits)
1360 {
1361 __fun(__countr_zero(__bits));
1362 __bits &= (__bits - 1);
1363 }
1364 }
1365
1366 /** @internal
1367 * Optimized @c memcpy for use in partial loads and stores.
1368 *
1369 * The implementation uses at most two fixed-size power-of-2 @c memcpy calls and reduces the
1370 * number of branches to a minimum. The variable size is achieved by overlapping two @c memcpy
1371 * calls.
1372 *
1373 * @tparam _Chunk Copies @p __n times @p _Chunk bytes.
1374 * @tparam _Max Copy no more than @p _Max bytes.
1375 *
1376 * @param __dst The destination pointer.
1377 * @param __src The source pointer.
1378 * @param __n Thu number of chunks that need to be copied.
1379 */
1380 template <size_t _Chunk, size_t _Max>
1381 inline void
1382 __memcpy_chunks(byte* __restrict__ __dst, const byte* __restrict__ __src,
1383 size_t __n)
1384 {
1385 static_assert(_Max <= 64);
1386 static_assert(__has_single_bit(_Chunk) && _Chunk <= 8);
1387 size_t __bytes = _Chunk * __n;
1388 if (__builtin_constant_p(__bytes))
1389 { // If __n is known via constant propagation use a single memcpy call. Since this is still
1390 // a fixed-size memcpy to the compiler, this leaves more room for optimization.
1391 __builtin_memcpy(__dst, __src, __bytes);
1392 }
1393 else if (__bytes > 32 && _Max > 32)
1394 {
1395 __builtin_memcpy(__dst, __src, 32);
1396 __bytes -= 32;
1397 __builtin_memcpy(__dst + __bytes, __src + __bytes, 32);
1398 }
1399 else if (__bytes > 16 && _Max > 16)
1400 {
1401 __builtin_memcpy(__dst, __src, 16);
1402 if constexpr (_Chunk == 8)
1403 {
1404 __bytes -= 8;
1405 __builtin_memcpy(__dst + __bytes, __src + __bytes, 8);
1406 }
1407 else
1408 {
1409 __bytes -= 16;
1410 __builtin_memcpy(__dst + __bytes, __src + __bytes, 16);
1411 }
1412 }
1413 else if (__bytes > 8 && _Max > 8)
1414 {
1415 __builtin_memcpy(__dst, __src, 8);
1416 if constexpr (_Chunk == 4)
1417 {
1418 __bytes -= 4;
1419 __builtin_memcpy(__dst + __bytes, __src + __bytes, 4);
1420 }
1421 else if constexpr (_Chunk < 4)
1422 {
1423 __bytes -= 8;
1424 __builtin_memcpy(__dst + __bytes, __src + __bytes, 8);
1425 }
1426 }
1427 else if (__bytes > 4 && _Max > 4)
1428 {
1429 __builtin_memcpy(__dst, __src, 4);
1430 if constexpr (_Chunk == 2)
1431 {
1432 __bytes -= 2;
1433 __builtin_memcpy(__dst + __bytes, __src + __bytes, 2);
1434 }
1435 else if constexpr (_Chunk == 1)
1436 {
1437 __bytes -= 4;
1438 __builtin_memcpy(__dst + __bytes, __src + __bytes, 4);
1439 }
1440 }
1441 else if (__bytes >= 2)
1442 {
1443 __builtin_memcpy(__dst, __src, 2);
1444 if constexpr (_Chunk == 2)
1445 {
1446 __bytes -= 2;
1447 __builtin_memcpy(__dst + __bytes, __src + __bytes, 2);
1448 }
1449 else if constexpr (_Chunk == 1)
1450 {
1451 __bytes -= 1;
1452 __builtin_memcpy(__dst + __bytes, __src + __bytes, 1);
1453 }
1454 }
1455 else if (__bytes == 1)
1456 __builtin_memcpy(__dst, __src, 1);
1457 }
1458
1459 // [simd.reductions] identity_element = *see below*
1460 template <typename _Tp, typename _BinaryOperation>
1461 requires __is_one_of<_BinaryOperation,
1462 plus<>, multiplies<>, bit_and<>, bit_or<>, bit_xor<>>::value
1463 consteval _Tp
1464 __default_identity_element()
1465 {
1466 if constexpr (same_as<_BinaryOperation, multiplies<>>)
1467 return _Tp(1);
1468 else if constexpr (same_as<_BinaryOperation, bit_and<>>)
1469 return _Tp(~_Tp());
1470 else
1471 return _Tp(0);
1472 }
1473} // namespace simd
1474_GLIBCXX_END_NAMESPACE_VERSION
1475} // namespace std
1476
1477#pragma GCC diagnostic pop
1478#endif // C++26
1479#endif // _GLIBCXX_SIMD_DETAILS_H
typename underlying_type< _Tp >::type underlying_type_t
Alias template for underlying_type.
Definition type_traits:2982
typename make_unsigned< _Tp >::type make_unsigned_t
Alias template for make_unsigned.
Definition type_traits:2274
auto declval() noexcept -> decltype(__declval< _Tp >(0))
Definition type_traits:2742
ISO C++ entities toplevel namespace is std.
__make_integer_seq< integer_sequence, _Tp, _Num > make_integer_sequence
Alias template make_integer_sequence.
Definition utility.h:517
__numeric_traits_integer< _Tp > __int_traits
Convenience alias for __numeric_traits<integer-type>.
static constexpr int digits
Definition limits:224
static constexpr _Tp max() noexcept
Definition limits:336
static constexpr _Tp lowest() noexcept
Definition limits:342