Eigen  5.0.1
 
Loading...
Searching...
No Matches
PacketMath.h
1// This file is part of Eigen, a lightweight C++ template library
2// for linear algebra.
3//
4// Copyright (C) 2025 Rasmus Munk Larsen
5//
6// This Source Code Form is subject to the terms of the Mozilla
7// Public License v. 2.0. If a copy of the MPL was not distributed
8// with this file, You can obtain one at http://mozilla.org/MPL/2.0/.
9// SPDX-License-Identifier: MPL-2.0
10
11#ifndef EIGEN_PACKET_MATH_CLANG_H
12#define EIGEN_PACKET_MATH_CLANG_H
13
14// IWYU pragma: private
15#include "../../InternalHeaderCheck.h"
16
17namespace Eigen {
18namespace internal {
19
20namespace detail {
21// namespace detail contains implementation details specific to this
22// file, while namespace internal contains internal APIs used elsewhere
23// in Eigen.
24template <typename ScalarT, int n>
25using VectorType = ScalarT __attribute__((ext_vector_type(n), aligned(n * sizeof(ScalarT))));
26} // namespace detail
27
28// --- Naming Convention ---
29// This backend uses size-independent type aliases so the same code works
30// for EIGEN_GENERIC_VECTOR_SIZE_BYTES in {16, 32, 64}:
31//
32// PacketXf - float vector (4, 8, or 16 elements)
33// PacketXd - double vector (2, 4, or 8 elements)
34// PacketXi - int32_t vector (4, 8, or 16 elements)
35// PacketXl - int64_t vector (2, 4, or 8 elements)
36// PacketXcf - complex<float> vector (2, 4, or 8 elements) [in Complex.h]
37// PacketXcd - complex<double> vector (1, 2, or 4 elements) [in Complex.h]
38//
39// The "X" suffix indicates the element count is determined by the macro
40// EIGEN_GENERIC_VECTOR_SIZE_BYTES at compile time. Operations that require
41// compile-time constant indices (e.g. __builtin_shufflevector) obtain them by
42// expanding detail::vector_indices<Packet>, so they need no per-size code.
43
44static_assert(EIGEN_GENERIC_VECTOR_SIZE_BYTES == 16 || EIGEN_GENERIC_VECTOR_SIZE_BYTES == 32 ||
45 EIGEN_GENERIC_VECTOR_SIZE_BYTES == 64,
46 "EIGEN_GENERIC_VECTOR_SIZE_BYTES must be 16, 32, or 64");
47
48constexpr int kFloatPacketSize = EIGEN_GENERIC_VECTOR_SIZE_BYTES / sizeof(float);
49constexpr int kDoublePacketSize = EIGEN_GENERIC_VECTOR_SIZE_BYTES / sizeof(double);
50using PacketXf = detail::VectorType<float, kFloatPacketSize>;
51using PacketXd = detail::VectorType<double, kDoublePacketSize>;
52using PacketXi = detail::VectorType<int32_t, kFloatPacketSize>;
53using PacketXl = detail::VectorType<int64_t, kDoublePacketSize>;
54
55// --- packet_traits specializations ---
56struct generic_float_packet_traits : default_packet_traits {
57 enum {
58 Vectorizable = 1,
59 AlignedOnScalar = 1,
60 HasAdd = 1,
61 HasSub = 1,
62 HasMul = 1,
63 HasDiv = 1,
64 HasNegate = 1,
65 HasAbs = 1,
66 HasRound = 1,
67 HasMin = 1,
68 HasMax = 1,
69 HasCmp = 1,
70 HasSet1 = 1,
71 HasCast = 1,
72 HasBitwise = 1,
73 HasRedux = 1,
74 HasSign = 1,
75 HasArg = 0,
76 HasConj = 1,
77 // Math functions
78 HasReciprocal = 1,
79 HasSin = 1,
80 HasCos = 1,
81 HasTan = 1,
82 HasACos = 1,
83 HasASin = 1,
84 HasATan = 1,
85 HasATanh = 1,
86 HasLog = 1,
87 HasLog1p = 1,
88 HasExpm1 = 1,
89 HasExp = 1,
90 HasPow = 1,
91 HasNdtri = 1,
92 HasBessel = 1,
93 HasSqrt = 1,
94 HasRsqrt = 1,
95 HasCbrt = 1,
96 HasTanh = 1,
97 HasErf = 1,
98 HasErfc = 1
99 };
100};
101
102template <>
103struct packet_traits<float> : generic_float_packet_traits {
104 using type = PacketXf;
105 using half = PacketXf;
106 enum {
107 size = kFloatPacketSize,
108 };
109};
110
111template <>
112struct packet_traits<double> : generic_float_packet_traits {
113 using type = PacketXd;
114 using half = PacketXd;
115 // Generic double-precision acos/asin are not yet implemented in
116 // GenericPacketMathFunctions.h (only float versions exist).
117 enum { size = kDoublePacketSize, HasACos = 0, HasASin = 0 };
118};
119
120struct generic_integer_packet_traits : default_packet_traits {
121 enum {
122 Vectorizable = 1,
123 AlignedOnScalar = 1,
124 HasAdd = 1,
125 HasSub = 1,
126 HasMul = 1,
127 HasDiv = 1,
128 HasNegate = 1,
129 HasAbs = 1,
130 HasMin = 1,
131 HasMax = 1,
132 HasCmp = 1,
133 HasSet1 = 1,
134 HasCast = 1,
135 HasBitwise = 1,
136 HasRedux = 1,
137 // Set remaining to 0
138 HasRound = 1,
139 HasSqrt = 0,
140 HasRsqrt = 0,
141 HasReciprocal = 0,
142 HasArg = 0,
143 HasConj = 1,
144 HasExp = 0,
145 HasLog = 0,
146 HasSin = 0,
147 HasCos = 0,
148 };
149};
150
151template <>
152struct packet_traits<int32_t> : generic_integer_packet_traits {
153 using type = PacketXi;
154 using half = PacketXi;
155 enum {
156 size = kFloatPacketSize,
157 };
158};
159
160template <>
161struct packet_traits<int64_t> : generic_integer_packet_traits {
162 using type = PacketXl;
163 using half = PacketXl;
164 enum {
165 size = kDoublePacketSize,
166 };
167};
168
169// --- unpacket_traits specializations ---
170struct generic_unpacket_traits : default_unpacket_traits {
171 enum {
172 alignment = EIGEN_GENERIC_VECTOR_SIZE_BYTES,
173 vectorizable = true,
174 };
175};
176
177template <>
178struct unpacket_traits<PacketXf> : generic_unpacket_traits {
179 using type = float;
180 using half = PacketXf;
181 using integer_packet = PacketXi;
182 enum {
183 size = kFloatPacketSize,
184 };
185};
186template <>
187struct unpacket_traits<PacketXd> : generic_unpacket_traits {
188 using type = double;
189 using half = PacketXd;
190 using integer_packet = PacketXl;
191 enum {
192 size = kDoublePacketSize,
193 };
194};
195template <>
196struct unpacket_traits<PacketXi> : generic_unpacket_traits {
197 using type = int32_t;
198 using half = PacketXi;
199 enum {
200 size = kFloatPacketSize,
201 };
202};
203template <>
204struct unpacket_traits<PacketXl> : generic_unpacket_traits {
205 using type = int64_t;
206 using half = PacketXl;
207 enum {
208 size = kDoublePacketSize,
209 };
210};
211
212namespace detail {
213// --- vector type helpers ---
214template <typename VectorT>
215struct ScalarTypeOfVector {
216 using type = std::remove_all_extents_t<std::remove_reference_t<decltype(VectorT()[0])>>;
217};
218
219template <typename VectorT>
220using scalar_type_of_vector_t = typename ScalarTypeOfVector<VectorT>::type;
221
222template <typename VectorType>
223struct UnsignedVectorHelper {
224 static VectorType v;
225 static constexpr int n = __builtin_vectorelements(v);
226 using UnsignedScalar = std::make_unsigned_t<scalar_type_of_vector_t<VectorType>>;
227 using type = UnsignedScalar __attribute__((ext_vector_type(n), aligned(n * sizeof(UnsignedScalar))));
228};
229
230template <typename VectorT>
231using unsigned_vector_t = typename UnsignedVectorHelper<VectorT>::type;
232
233template <typename VectorT>
234constexpr int vector_elements() {
235 return static_cast<int>(sizeof(VectorT) / sizeof(scalar_type_of_vector_t<VectorT>));
236}
237
238// Signed integer vector with the same lane count and width, for sign-bit
239// tests and bitwise manipulation of floating-point packets.
240template <typename VectorT>
241struct SignedVectorHelper {
242 using SignedScalar = std::conditional_t<sizeof(scalar_type_of_vector_t<VectorT>) == 4, int32_t, int64_t>;
243 using type = VectorType<SignedScalar, vector_elements<VectorT>()>;
244};
245
246template <typename VectorT>
247using signed_vector_t = typename SignedVectorHelper<VectorT>::type;
248
249template <typename VectorT>
250using half_vector_t = VectorType<scalar_type_of_vector_t<VectorT>, vector_elements<VectorT>() / 2>;
251
252template <typename VectorT>
253using quarter_vector_t = VectorType<scalar_type_of_vector_t<VectorT>, vector_elements<VectorT>() / 4>;
254
255template <typename VectorT>
256using scalar_pair_t = std::pair<scalar_type_of_vector_t<VectorT>, scalar_type_of_vector_t<VectorT>>;
257
258// Index sequence covering every element of VectorT. Expanding it inside a
259// __builtin_shufflevector index list or a braced initializer is what keeps the
260// operations below independent of EIGEN_GENERIC_VECTOR_SIZE_BYTES.
261template <typename VectorT>
262using vector_indices = std::make_index_sequence<vector_elements<VectorT>()>;
263
264// load and store helpers.
265template <typename VectorT>
266EIGEN_STRONG_INLINE VectorT load_vector_unaligned(const scalar_type_of_vector_t<VectorT>* from) {
267 VectorT to;
268 __builtin_memcpy(&to, from, sizeof(VectorT));
269 return to;
270}
271
272template <typename VectorT>
273EIGEN_STRONG_INLINE VectorT load_vector_aligned(const scalar_type_of_vector_t<VectorT>* from) {
274 eigen_assert((std::uintptr_t(from) % alignof(VectorT) == 0) && "load_vector_aligned");
275 return *reinterpret_cast<const VectorT*>(assume_aligned<alignof(VectorT)>(from));
276}
277
278template <typename VectorT>
279EIGEN_STRONG_INLINE void store_vector_unaligned(scalar_type_of_vector_t<VectorT>* to, const VectorT& from) {
280 __builtin_memcpy(to, &from, sizeof(VectorT));
281}
282
283template <typename VectorT>
284EIGEN_STRONG_INLINE void store_vector_aligned(scalar_type_of_vector_t<VectorT>* to, const VectorT& from) {
285 eigen_assert((std::uintptr_t(to) % alignof(VectorT) == 0) && "store_vector_aligned");
286 *reinterpret_cast<VectorT*>(assume_aligned<alignof(VectorT)>(to)) = from;
287}
288
289} // namespace detail
290
291// --- Intrinsic-like specializations ---
292
293// --- Load/Store operations ---
294#define EIGEN_CLANG_PACKET_LOAD_STORE_PACKET(PACKET_TYPE) \
295 template <> \
296 EIGEN_STRONG_INLINE PACKET_TYPE ploadu<PACKET_TYPE>(const detail::scalar_type_of_vector_t<PACKET_TYPE>* from) { \
297 return detail::load_vector_unaligned<PACKET_TYPE>(from); \
298 } \
299 template <> \
300 EIGEN_STRONG_INLINE PACKET_TYPE pload<PACKET_TYPE>(const detail::scalar_type_of_vector_t<PACKET_TYPE>* from) { \
301 return detail::load_vector_aligned<PACKET_TYPE>(from); \
302 } \
303 template <> \
304 EIGEN_STRONG_INLINE void pstoreu<detail::scalar_type_of_vector_t<PACKET_TYPE>, PACKET_TYPE>( \
305 detail::scalar_type_of_vector_t<PACKET_TYPE> * to, const PACKET_TYPE& from) { \
306 detail::store_vector_unaligned<PACKET_TYPE>(to, from); \
307 } \
308 template <> \
309 EIGEN_STRONG_INLINE void pstore<detail::scalar_type_of_vector_t<PACKET_TYPE>, PACKET_TYPE>( \
310 detail::scalar_type_of_vector_t<PACKET_TYPE> * to, const PACKET_TYPE& from) { \
311 detail::store_vector_aligned<PACKET_TYPE>(to, from); \
312 }
313
314EIGEN_CLANG_PACKET_LOAD_STORE_PACKET(PacketXf)
315EIGEN_CLANG_PACKET_LOAD_STORE_PACKET(PacketXd)
316EIGEN_CLANG_PACKET_LOAD_STORE_PACKET(PacketXi)
317EIGEN_CLANG_PACKET_LOAD_STORE_PACKET(PacketXl)
318#undef EIGEN_CLANG_PACKET_LOAD_STORE_PACKET
319
320// --- Broadcast operation ---
321template <>
322EIGEN_STRONG_INLINE PacketXf pset1frombits<PacketXf>(uint32_t from) {
323 return PacketXf(numext::bit_cast<float>(from));
324}
325
326template <>
327EIGEN_STRONG_INLINE PacketXd pset1frombits<PacketXd>(uint64_t from) {
328 return PacketXd(numext::bit_cast<double>(from));
329}
330
331#define EIGEN_CLANG_PACKET_SET1(PACKET_TYPE) \
332 template <> \
333 EIGEN_STRONG_INLINE PACKET_TYPE pset1<PACKET_TYPE>(const unpacket_traits<PACKET_TYPE>::type& from) { \
334 return PACKET_TYPE(from); \
335 } \
336 template <> \
337 EIGEN_STRONG_INLINE unpacket_traits<PACKET_TYPE>::type pfirst<PACKET_TYPE>(const PACKET_TYPE& from) { \
338 return from[0]; \
339 }
340
341EIGEN_CLANG_PACKET_SET1(PacketXf)
342EIGEN_CLANG_PACKET_SET1(PacketXd)
343EIGEN_CLANG_PACKET_SET1(PacketXi)
344EIGEN_CLANG_PACKET_SET1(PacketXl)
345#undef EIGEN_CLANG_PACKET_SET1
346
347// --- Arithmetic operations ---
348#define EIGEN_CLANG_PACKET_ARITHMETIC(PACKET_TYPE) \
349 template <> \
350 EIGEN_STRONG_INLINE PACKET_TYPE pisnan<PACKET_TYPE>(const PACKET_TYPE& a) { \
351 return reinterpret_cast<PACKET_TYPE>(a != a); \
352 } \
353 template <> \
354 EIGEN_STRONG_INLINE PACKET_TYPE pnegate<PACKET_TYPE>(const PACKET_TYPE& a) { \
355 return -a; \
356 }
357
358EIGEN_CLANG_PACKET_ARITHMETIC(PacketXf)
359EIGEN_CLANG_PACKET_ARITHMETIC(PacketXd)
360EIGEN_CLANG_PACKET_ARITHMETIC(PacketXi)
361EIGEN_CLANG_PACKET_ARITHMETIC(PacketXl)
362#undef EIGEN_CLANG_PACKET_ARITHMETIC
363
364// --- Bitwise operations (via casting) ---
365
366namespace detail {
367
368// Reinterpret-cast helpers, equivalent to preinterpret<> but defined here
369// because PacketMath.h is included before TypeCasting.h.
370EIGEN_STRONG_INLINE PacketXi preinterpret_float_to_int(const PacketXf& a) { return reinterpret_cast<PacketXi>(a); }
371EIGEN_STRONG_INLINE PacketXf preinterpret_int_to_float(const PacketXi& a) { return reinterpret_cast<PacketXf>(a); }
372EIGEN_STRONG_INLINE PacketXl preinterpret_double_to_long(const PacketXd& a) { return reinterpret_cast<PacketXl>(a); }
373EIGEN_STRONG_INLINE PacketXd preinterpret_long_to_double(const PacketXl& a) { return reinterpret_cast<PacketXd>(a); }
374
375} // namespace detail
376
377// Bitwise ops for integer packets
378#define EIGEN_CLANG_PACKET_BITWISE_INT(PACKET_TYPE) \
379 template <> \
380 constexpr EIGEN_STRONG_INLINE PACKET_TYPE pzero<PACKET_TYPE>(const PACKET_TYPE& /*unused*/) { \
381 return PACKET_TYPE(0); \
382 } \
383 template <> \
384 constexpr EIGEN_STRONG_INLINE PACKET_TYPE ptrue<PACKET_TYPE>(const PACKET_TYPE& /*unused*/) { \
385 return numext::bit_cast<PACKET_TYPE>(PACKET_TYPE(0) == PACKET_TYPE(0)); \
386 } \
387 template <> \
388 EIGEN_STRONG_INLINE PACKET_TYPE pand<PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b) { \
389 return a & b; \
390 } \
391 template <> \
392 EIGEN_STRONG_INLINE PACKET_TYPE por<PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b) { \
393 return a | b; \
394 } \
395 template <> \
396 EIGEN_STRONG_INLINE PACKET_TYPE pxor<PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b) { \
397 return a ^ b; \
398 } \
399 template <> \
400 EIGEN_STRONG_INLINE PACKET_TYPE pandnot<PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b) { \
401 return a & ~b; \
402 } \
403 template <int N> \
404 EIGEN_STRONG_INLINE PACKET_TYPE parithmetic_shift_right(const PACKET_TYPE& a) { \
405 return a >> N; \
406 } \
407 template <int N> \
408 EIGEN_STRONG_INLINE PACKET_TYPE plogical_shift_right(const PACKET_TYPE& a) { \
409 using UnsignedT = detail::unsigned_vector_t<PACKET_TYPE>; \
410 return reinterpret_cast<PACKET_TYPE>(reinterpret_cast<UnsignedT>(a) >> N); \
411 } \
412 template <int N> \
413 EIGEN_STRONG_INLINE PACKET_TYPE plogical_shift_left(const PACKET_TYPE& a) { \
414 return a << N; \
415 }
416
417EIGEN_CLANG_PACKET_BITWISE_INT(PacketXi)
418EIGEN_CLANG_PACKET_BITWISE_INT(PacketXl)
419#undef EIGEN_CLANG_PACKET_BITWISE_INT
420
421// Bitwise ops for floating point packets
422#define EIGEN_CLANG_PACKET_BITWISE_FLOAT(PACKET_TYPE, CAST_TO_INT, CAST_FROM_INT) \
423 template <> \
424 constexpr EIGEN_STRONG_INLINE PACKET_TYPE pzero<PACKET_TYPE>(const PACKET_TYPE& /*unused*/) { \
425 using Scalar = detail::scalar_type_of_vector_t<PACKET_TYPE>; \
426 return PACKET_TYPE(Scalar(0)); \
427 } \
428 template <> \
429 EIGEN_STRONG_INLINE PACKET_TYPE ptrue<PACKET_TYPE>(const PACKET_TYPE& /* unused */) { \
430 using Scalar = detail::scalar_type_of_vector_t<PACKET_TYPE>; \
431 PACKET_TYPE r = numext::bit_cast<PACKET_TYPE>(PACKET_TYPE(Scalar(0)) == PACKET_TYPE(Scalar(0))); \
432 EIGEN_FAST_MATH_CONSTANT_BARRIER(r); \
433 return r; \
434 } \
435 template <> \
436 EIGEN_STRONG_INLINE PACKET_TYPE pand<PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b) { \
437 return CAST_FROM_INT(CAST_TO_INT(a) & CAST_TO_INT(b)); \
438 } \
439 template <> \
440 EIGEN_STRONG_INLINE PACKET_TYPE por<PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b) { \
441 return CAST_FROM_INT(CAST_TO_INT(a) | CAST_TO_INT(b)); \
442 } \
443 template <> \
444 EIGEN_STRONG_INLINE PACKET_TYPE pxor<PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b) { \
445 return CAST_FROM_INT(CAST_TO_INT(a) ^ CAST_TO_INT(b)); \
446 } \
447 template <> \
448 EIGEN_STRONG_INLINE PACKET_TYPE pandnot<PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b) { \
449 return CAST_FROM_INT(CAST_TO_INT(a) & ~CAST_TO_INT(b)); \
450 }
451
452EIGEN_CLANG_PACKET_BITWISE_FLOAT(PacketXf, detail::preinterpret_float_to_int, detail::preinterpret_int_to_float)
453EIGEN_CLANG_PACKET_BITWISE_FLOAT(PacketXd, detail::preinterpret_double_to_long, detail::preinterpret_long_to_double)
454#undef EIGEN_CLANG_PACKET_BITWISE_FLOAT
455
456// --- Comparison operations ---
457// Clang vector extensions perform comparisons in the original type (float/double),
458// returning an int vector with all-ones (-1) for true and all-zeros for false.
459// The bit_cast reinterprets those int bitmasks as float packets, which is the
460// format expected by pselect and other Eigen packet operations.
461#define EIGEN_CLANG_PACKET_CMP(PACKET_TYPE, INT_PACKET_TYPE) \
462 template <> \
463 EIGEN_STRONG_INLINE PACKET_TYPE pcmp_eq<PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b) { \
464 return numext::bit_cast<PACKET_TYPE>(INT_PACKET_TYPE(a == b)); \
465 } \
466 template <> \
467 EIGEN_STRONG_INLINE PACKET_TYPE pcmp_lt<PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b) { \
468 return numext::bit_cast<PACKET_TYPE>(INT_PACKET_TYPE(a < b)); \
469 } \
470 template <> \
471 EIGEN_STRONG_INLINE PACKET_TYPE pcmp_le<PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b) { \
472 return numext::bit_cast<PACKET_TYPE>(INT_PACKET_TYPE(a <= b)); \
473 } \
474 template <> \
475 EIGEN_STRONG_INLINE PACKET_TYPE pcmp_lt_or_nan<PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b) { \
476 return numext::bit_cast<PACKET_TYPE>(INT_PACKET_TYPE(!(a >= b))); \
477 }
478
479EIGEN_CLANG_PACKET_CMP(PacketXf, PacketXi)
480EIGEN_CLANG_PACKET_CMP(PacketXd, PacketXl)
481#undef EIGEN_CLANG_PACKET_CMP
482
483// --- Min/Max/select operations ---
484namespace detail {
485// Functors usable at any vector width; the min/max reduction trees in
486// Reductions.h reuse them on progressively narrower vectors. The
487// compare-select forms compile to a single min/max instruction on targets
488// whose min/max returns the second operand when the inputs are unordered
489// (e.g. x86), and they spell out the NaN propagation of std::min/std::max:
490// the first argument is returned if either input is NaN.
491struct pmin_op {
492 template <typename VectorT>
493 EIGEN_STRONG_INLINE VectorT operator()(const VectorT& a, const VectorT& b) const {
494 return b < a ? b : a;
495 }
496};
497struct pmax_op {
498 template <typename VectorT>
499 EIGEN_STRONG_INLINE VectorT operator()(const VectorT& a, const VectorT& b) const {
500 return b > a ? b : a;
501 }
502};
503// IEEE 754-2008 minNum/maxNum semantics: return the other operand if one input
504// is NaN. Floating-point support in __builtin_elementwise_{min,max} is
505// deprecated because the name does not say which of the several IEEE min/max
506// flavors is meant; __builtin_elementwise_{minnum,maxnum} spell out the same
507// semantics the deprecated builtins provided for floats, and additionally pin
508// down +0.0 > -0.0. The elementwise_{min,max} fallback (always available
509// under this backend's clang >= 16 gate) has the same NaN semantics but
510// leaves the zero-sign tie unspecified.
511struct pmin_num_op {
512 template <typename VectorT>
513 EIGEN_STRONG_INLINE VectorT operator()(const VectorT& a, const VectorT& b) const {
514#if EIGEN_HAS_BUILTIN(__builtin_elementwise_minnum)
515 return __builtin_elementwise_minnum(a, b);
516#else
517 return __builtin_elementwise_min(a, b);
518#endif
519 }
520};
521struct pmax_num_op {
522 template <typename VectorT>
523 EIGEN_STRONG_INLINE VectorT operator()(const VectorT& a, const VectorT& b) const {
524#if EIGEN_HAS_BUILTIN(__builtin_elementwise_maxnum)
525 return __builtin_elementwise_maxnum(a, b);
526#else
527 return __builtin_elementwise_max(a, b);
528#endif
529 }
530};
531// Return NaN if either input is NaN, otherwise the min/max. When a is NaN the
532// plain compare-select form already returns a, so only b needs an explicit
533// test.
534struct pmin_nan_op {
535 template <typename VectorT>
536 EIGEN_STRONG_INLINE VectorT operator()(const VectorT& a, const VectorT& b) const {
537 return b != b ? b : pmin_op()(a, b);
538 }
539};
540struct pmax_nan_op {
541 template <typename VectorT>
542 EIGEN_STRONG_INLINE VectorT operator()(const VectorT& a, const VectorT& b) const {
543 return b != b ? b : pmax_op()(a, b);
544 }
545};
546} // namespace detail
547
548// pmin/pmax/pselect are pure compare-select code and apply to all packet types.
549#define EIGEN_CLANG_PACKET_MINMAX_SELECT(PACKET_TYPE) \
550 template <> \
551 EIGEN_STRONG_INLINE PACKET_TYPE pmin<PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b) { \
552 return detail::pmin_op()(a, b); \
553 } \
554 template <> \
555 EIGEN_STRONG_INLINE PACKET_TYPE pmax<PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b) { \
556 return detail::pmax_op()(a, b); \
557 } \
558 template <> \
559 EIGEN_STRONG_INLINE PACKET_TYPE pselect<PACKET_TYPE>(const PACKET_TYPE& mask, const PACKET_TYPE& a, \
560 const PACKET_TYPE& b) { \
561 /* The mask is all-ones or all-zeros per lane, so testing the sign of the */ \
562 /* signed integer view suffices and maps to a single blend instruction. */ \
563 /* Unlike a floating-point `mask != 0` test it also survives -ffast-math, */ \
564 /* which may assume the all-ones NaN bit pattern cannot occur in a float. */ \
565 return reinterpret_cast<detail::signed_vector_t<PACKET_TYPE>>(mask) < 0 ? a : b; \
566 }
567
568EIGEN_CLANG_PACKET_MINMAX_SELECT(PacketXf)
569EIGEN_CLANG_PACKET_MINMAX_SELECT(PacketXd)
570EIGEN_CLANG_PACKET_MINMAX_SELECT(PacketXi)
571EIGEN_CLANG_PACKET_MINMAX_SELECT(PacketXl)
572#undef EIGEN_CLANG_PACKET_MINMAX_SELECT
573
574// NaN-propagation variants for the floating-point packets.
575#define EIGEN_CLANG_PACKET_MINMAX_FLOAT(PACKET_TYPE) \
576 template <> \
577 EIGEN_STRONG_INLINE PACKET_TYPE pmin<PropagateNumbers, PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b) { \
578 return detail::pmin_num_op()(a, b); \
579 } \
580 template <> \
581 EIGEN_STRONG_INLINE PACKET_TYPE pmax<PropagateNumbers, PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b) { \
582 return detail::pmax_num_op()(a, b); \
583 } \
584 template <> \
585 EIGEN_STRONG_INLINE PACKET_TYPE pmin<PropagateNaN, PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b) { \
586 return detail::pmin_nan_op()(a, b); \
587 } \
588 template <> \
589 EIGEN_STRONG_INLINE PACKET_TYPE pmax<PropagateNaN, PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b) { \
590 return detail::pmax_nan_op()(a, b); \
591 }
592
593EIGEN_CLANG_PACKET_MINMAX_FLOAT(PacketXf)
594EIGEN_CLANG_PACKET_MINMAX_FLOAT(PacketXd)
595#undef EIGEN_CLANG_PACKET_MINMAX_FLOAT
596
597#if EIGEN_HAS_BUILTIN(__builtin_elementwise_abs)
598#define EIGEN_CLANG_PACKET_ABS(PACKET_TYPE) \
599 template <> \
600 EIGEN_STRONG_INLINE PACKET_TYPE pabs<PACKET_TYPE>(const PACKET_TYPE& a) { \
601 return __builtin_elementwise_abs(a); \
602 }
603
604EIGEN_CLANG_PACKET_ABS(PacketXf)
605EIGEN_CLANG_PACKET_ABS(PacketXd)
606EIGEN_CLANG_PACKET_ABS(PacketXi)
607EIGEN_CLANG_PACKET_ABS(PacketXl)
608#undef EIGEN_CLANG_PACKET_ABS
609#endif
610
611// psignbit: a signed compare of the integer view is a single instruction,
612// unlike the generic floating-point fallback.
613template <>
614EIGEN_STRONG_INLINE PacketXf psignbit(const PacketXf& a) {
615 return reinterpret_cast<PacketXf>(reinterpret_cast<PacketXi>(a) < 0);
616}
617template <>
618EIGEN_STRONG_INLINE PacketXd psignbit(const PacketXd& a) {
619 return reinterpret_cast<PacketXd>(reinterpret_cast<PacketXl>(a) < 0);
620}
621
622// --- Math functions (float/double only) ---
623
624#if EIGEN_HAS_BUILTIN(__builtin_elementwise_floor) && EIGEN_HAS_BUILTIN(__builtin_elementwise_ceil) && \
625 EIGEN_HAS_BUILTIN(__builtin_elementwise_round) && EIGEN_HAS_BUILTIN(__builtin_elementwise_roundeven) && \
626 EIGEN_HAS_BUILTIN(__builtin_elementwise_trunc) && EIGEN_HAS_BUILTIN(__builtin_elementwise_sqrt)
627#define EIGEN_CLANG_PACKET_MATH_FLOAT(PACKET_TYPE) \
628 template <> \
629 EIGEN_STRONG_INLINE PACKET_TYPE pfloor<PACKET_TYPE>(const PACKET_TYPE& a) { \
630 return __builtin_elementwise_floor(a); \
631 } \
632 template <> \
633 EIGEN_STRONG_INLINE PACKET_TYPE pceil<PACKET_TYPE>(const PACKET_TYPE& a) { \
634 return __builtin_elementwise_ceil(a); \
635 } \
636 template <> \
637 EIGEN_STRONG_INLINE PACKET_TYPE pround<PACKET_TYPE>(const PACKET_TYPE& a) { \
638 return __builtin_elementwise_round(a); \
639 } \
640 template <> \
641 EIGEN_STRONG_INLINE PACKET_TYPE print<PACKET_TYPE>(const PACKET_TYPE& a) { \
642 return __builtin_elementwise_roundeven(a); \
643 } \
644 template <> \
645 EIGEN_STRONG_INLINE PACKET_TYPE ptrunc<PACKET_TYPE>(const PACKET_TYPE& a) { \
646 return __builtin_elementwise_trunc(a); \
647 } \
648 template <> \
649 EIGEN_STRONG_INLINE PACKET_TYPE psqrt<PACKET_TYPE>(const PACKET_TYPE& a) { \
650 return __builtin_elementwise_sqrt(a); \
651 }
652
653EIGEN_CLANG_PACKET_MATH_FLOAT(PacketXf)
654EIGEN_CLANG_PACKET_MATH_FLOAT(PacketXd)
655#undef EIGEN_CLANG_PACKET_MATH_FLOAT
656#endif
657
658// --- Fused Multiply-Add (MADD) ---
659#if (defined(EIGEN_VECTORIZE_FMA) || defined(__FMA__)) && EIGEN_HAS_BUILTIN(__builtin_elementwise_fma)
660#define EIGEN_CLANG_PACKET_MADD(PACKET_TYPE) \
661 template <> \
662 EIGEN_STRONG_INLINE PACKET_TYPE pmadd<PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b, \
663 const PACKET_TYPE& c) { \
664 return __builtin_elementwise_fma(a, b, c); \
665 } \
666 template <> \
667 EIGEN_STRONG_INLINE PACKET_TYPE pmsub<PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b, \
668 const PACKET_TYPE& c) { \
669 return __builtin_elementwise_fma(a, b, -c); \
670 } \
671 template <> \
672 EIGEN_STRONG_INLINE PACKET_TYPE pnmadd<PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b, \
673 const PACKET_TYPE& c) { \
674 return __builtin_elementwise_fma(-a, b, c); \
675 } \
676 template <> \
677 EIGEN_STRONG_INLINE PACKET_TYPE pnmsub<PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b, \
678 const PACKET_TYPE& c) { \
679 return -(__builtin_elementwise_fma(a, b, c)); \
680 }
681#else
682// Fallback if FMA builtin is not available
683#define EIGEN_CLANG_PACKET_MADD(PACKET_TYPE) \
684 template <> \
685 EIGEN_STRONG_INLINE PACKET_TYPE pmadd<PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b, \
686 const PACKET_TYPE& c) { \
687 return (a * b) + c; \
688 } \
689 template <> \
690 EIGEN_STRONG_INLINE PACKET_TYPE pmsub<PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b, \
691 const PACKET_TYPE& c) { \
692 return (a * b) - c; \
693 } \
694 template <> \
695 EIGEN_STRONG_INLINE PACKET_TYPE pnmadd<PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b, \
696 const PACKET_TYPE& c) { \
697 return c - (a * b); \
698 } \
699 template <> \
700 EIGEN_STRONG_INLINE PACKET_TYPE pnmsub<PACKET_TYPE>(const PACKET_TYPE& a, const PACKET_TYPE& b, \
701 const PACKET_TYPE& c) { \
702 return -((a * b) + c); \
703 }
704#endif
705
706EIGEN_CLANG_PACKET_MADD(PacketXf)
707EIGEN_CLANG_PACKET_MADD(PacketXd)
708#undef EIGEN_CLANG_PACKET_MADD
709
710#define EIGEN_CLANG_PACKET_SCATTER_GATHER(PACKET_TYPE) \
711 template <> \
712 EIGEN_STRONG_INLINE void pscatter(unpacket_traits<PACKET_TYPE>::type* to, const PACKET_TYPE& from, Index stride) { \
713 constexpr int size = unpacket_traits<PACKET_TYPE>::size; \
714 for (int i = 0; i < size; ++i) { \
715 to[i * stride] = from[i]; \
716 } \
717 } \
718 template <> \
719 EIGEN_STRONG_INLINE PACKET_TYPE pgather<typename unpacket_traits<PACKET_TYPE>::type, PACKET_TYPE>( \
720 const unpacket_traits<PACKET_TYPE>::type* from, Index stride) { \
721 constexpr int size = unpacket_traits<PACKET_TYPE>::size; \
722 PACKET_TYPE result; \
723 for (int i = 0; i < size; ++i) { \
724 result[i] = from[i * stride]; \
725 } \
726 return result; \
727 }
728
729EIGEN_CLANG_PACKET_SCATTER_GATHER(PacketXf)
730EIGEN_CLANG_PACKET_SCATTER_GATHER(PacketXd)
731EIGEN_CLANG_PACKET_SCATTER_GATHER(PacketXi)
732EIGEN_CLANG_PACKET_SCATTER_GATHER(PacketXl)
733
734#undef EIGEN_CLANG_PACKET_SCATTER_GATHER
735
736// ---- Various operations that depend on __builtin_shufflevector.
737#if EIGEN_HAS_BUILTIN(__builtin_shufflevector)
738namespace detail {
739
740// --- Half / whole vector helpers ---
741template <typename VectorT, std::size_t... Is>
742EIGEN_STRONG_INLINE half_vector_t<VectorT> lower_half_impl(const VectorT& a, std::index_sequence<Is...>) {
743 return __builtin_shufflevector(a, a, Is...);
744}
745
746template <typename VectorT, std::size_t... Is>
747EIGEN_STRONG_INLINE half_vector_t<VectorT> upper_half_impl(const VectorT& a, std::index_sequence<Is...>) {
748 return __builtin_shufflevector(a, a, (sizeof...(Is) + Is)...);
749}
750
751template <typename VectorT, typename HalfT, std::size_t... Is>
752EIGEN_STRONG_INLINE VectorT concat_halves_impl(const HalfT& lo, const HalfT& hi, std::index_sequence<Is...>) {
753 return __builtin_shufflevector(lo, hi, Is...);
754}
755
756template <typename VectorT>
757EIGEN_STRONG_INLINE half_vector_t<VectorT> lower_half(const VectorT& a) {
758 return lower_half_impl(a, vector_indices<half_vector_t<VectorT>>{});
759}
760
761template <typename VectorT>
762EIGEN_STRONG_INLINE half_vector_t<VectorT> upper_half(const VectorT& a) {
763 return upper_half_impl(a, vector_indices<half_vector_t<VectorT>>{});
764}
765
766template <typename VectorT, typename HalfT>
767EIGEN_STRONG_INLINE VectorT concat_halves(const HalfT& lo, const HalfT& hi) {
768 return concat_halves_impl<VectorT>(lo, hi, vector_indices<VectorT>{});
769}
770
771// --- Width-generic bodies for the packet operations below ---
772template <typename Packet, std::size_t... Is>
773EIGEN_STRONG_INLINE Packet preverse_impl(const Packet& a, std::index_sequence<Is...>) {
774 return __builtin_shufflevector(a, a, (sizeof...(Is) - 1 - Is)...);
775}
776
777// Loads half a packet worth of scalars and repeats each of them twice.
778template <typename Packet, std::size_t... Is>
779EIGEN_STRONG_INLINE Packet ploaddup_impl(const typename unpacket_traits<Packet>::type* from,
780 std::index_sequence<Is...>) {
781 static_assert((unpacket_traits<Packet>::size) % 2 == 0, "Packet size must be a multiple of 2");
782 using HalfT = half_vector_t<Packet>;
783 const HalfT a = load_vector_unaligned<HalfT>(from);
784 return __builtin_shufflevector(a, a, (Is / 2)...);
785}
786
787// Loads a quarter of a packet worth of scalars and repeats each of them four times.
788template <typename Packet, std::size_t... Is>
789EIGEN_STRONG_INLINE Packet ploadquad_impl(const typename unpacket_traits<Packet>::type* from,
790 std::index_sequence<Is...>) {
791 static_assert((unpacket_traits<Packet>::size) % 4 == 0, "Packet size must be a multiple of 4");
792 using QuarterT = quarter_vector_t<Packet>;
793 const QuarterT a = load_vector_unaligned<QuarterT>(from);
794 return __builtin_shufflevector(a, a, (Is / 4)...);
795}
796
797template <typename Packet, std::size_t... Is>
798EIGEN_STRONG_INLINE Packet plset_impl(const typename unpacket_traits<Packet>::type& a, std::index_sequence<Is...>) {
799 using Scalar = typename unpacket_traits<Packet>::type;
800 return Packet{(a + Scalar(Is))...};
801}
802
803// All ones in the even lanes, all zeros in the odd ones. Return the integer representation so finite fast-math cannot
804// make the all-ones lanes poison before the caller applies EIGEN_FAST_MATH_CONSTANT_BARRIER.
805template <typename Packet, std::size_t... Is>
806EIGEN_STRONG_INLINE typename unpacket_traits<Packet>::integer_packet peven_mask_impl(std::index_sequence<Is...>) {
807 using IntegerPacket = typename unpacket_traits<Packet>::integer_packet;
808 using Bits = scalar_type_of_vector_t<IntegerPacket>;
809 return IntegerPacket{(Is % 2 == 0 ? Bits(-1) : Bits(0))...};
810}
811
812} // namespace detail
813
814#define EIGEN_CLANG_PACKET_PREVERSE(PACKET_TYPE) \
815 template <> \
816 EIGEN_STRONG_INLINE PACKET_TYPE preverse<PACKET_TYPE>(const PACKET_TYPE& a) { \
817 return detail::preverse_impl(a, detail::vector_indices<PACKET_TYPE>{}); \
818 }
819
820EIGEN_CLANG_PACKET_PREVERSE(PacketXf)
821EIGEN_CLANG_PACKET_PREVERSE(PacketXd)
822EIGEN_CLANG_PACKET_PREVERSE(PacketXi)
823EIGEN_CLANG_PACKET_PREVERSE(PacketXl)
824#undef EIGEN_CLANG_PACKET_PREVERSE
825
826#define EIGEN_CLANG_PACKET_LOADDUP(PACKET_TYPE) \
827 template <> \
828 EIGEN_STRONG_INLINE PACKET_TYPE ploaddup<PACKET_TYPE>(const unpacket_traits<PACKET_TYPE>::type* from) { \
829 return detail::ploaddup_impl<PACKET_TYPE>(from, detail::vector_indices<PACKET_TYPE>{}); \
830 }
831
832EIGEN_CLANG_PACKET_LOADDUP(PacketXf)
833EIGEN_CLANG_PACKET_LOADDUP(PacketXd)
834EIGEN_CLANG_PACKET_LOADDUP(PacketXi)
835EIGEN_CLANG_PACKET_LOADDUP(PacketXl)
836#undef EIGEN_CLANG_PACKET_LOADDUP
837
838#define EIGEN_CLANG_PACKET_LOADQUAD(PACKET_TYPE) \
839 template <> \
840 EIGEN_STRONG_INLINE PACKET_TYPE ploadquad<PACKET_TYPE>(const unpacket_traits<PACKET_TYPE>::type* from) { \
841 return detail::ploadquad_impl<PACKET_TYPE>(from, detail::vector_indices<PACKET_TYPE>{}); \
842 }
843
844EIGEN_CLANG_PACKET_LOADQUAD(PacketXf)
845EIGEN_CLANG_PACKET_LOADQUAD(PacketXi)
846#if EIGEN_GENERIC_VECTOR_SIZE_BYTES >= 32
847// PacketXd and PacketXl hold only two elements at 16 bytes, so they have no quarter packet to load from.
848EIGEN_CLANG_PACKET_LOADQUAD(PacketXd)
849EIGEN_CLANG_PACKET_LOADQUAD(PacketXl)
850#endif
851#undef EIGEN_CLANG_PACKET_LOADQUAD
852
853#define EIGEN_CLANG_PACKET_PLSET(PACKET_TYPE) \
854 template <> \
855 EIGEN_STRONG_INLINE PACKET_TYPE plset<PACKET_TYPE>(const unpacket_traits<PACKET_TYPE>::type& a) { \
856 return detail::plset_impl<PACKET_TYPE>(a, detail::vector_indices<PACKET_TYPE>{}); \
857 }
858
859EIGEN_CLANG_PACKET_PLSET(PacketXf)
860EIGEN_CLANG_PACKET_PLSET(PacketXd)
861EIGEN_CLANG_PACKET_PLSET(PacketXi)
862EIGEN_CLANG_PACKET_PLSET(PacketXl)
863#undef EIGEN_CLANG_PACKET_PLSET
864
865// --- peven_mask ---
866template <>
867EIGEN_STRONG_INLINE PacketXf peven_mask(const PacketXf& /* unused */) {
868 PacketXf r = numext::bit_cast<PacketXf>(detail::peven_mask_impl<PacketXf>(detail::vector_indices<PacketXf>{}));
869 EIGEN_FAST_MATH_CONSTANT_BARRIER(r);
870 return r;
871}
872template <>
873EIGEN_STRONG_INLINE PacketXd peven_mask(const PacketXd& /* unused */) {
874 PacketXd r = numext::bit_cast<PacketXd>(detail::peven_mask_impl<PacketXd>(detail::vector_indices<PacketXd>{}));
875 EIGEN_FAST_MATH_CONSTANT_BARRIER(r);
876 return r;
877}
878
879// Helpers for ptranspose.
880namespace detail {
881
882// Shuffle index of output element `i` when interleaving two vectors of `Size`
883// elements each. `Group` adjacent elements move together: 1 for scalar packets,
884// 2 for the complex packets in Complex.h, whose real and imaginary parts must
885// stay adjacent. Output groups alternate between the two inputs, taking group
886// `first_group` of each first.
887template <std::size_t Group, std::size_t Size>
888constexpr std::size_t zip_index(std::size_t i, std::size_t first_group) {
889 return Size * ((i / Group) % 2) + Group * (first_group + i / Group / 2) + i % Group;
890}
891
892// Interleaves p1 and p2 in place, leaving the low half of the result in p1 and
893// the high half in p2.
894template <std::size_t Group, typename VectorT, std::size_t... Is>
895EIGEN_ALWAYS_INLINE void zip_in_place_impl(VectorT& p1, VectorT& p2, std::index_sequence<Is...>) {
896 constexpr std::size_t kSize = sizeof...(Is);
897 // With a single lane group per vector both output shuffles would pick group
898 // 0 and silently duplicate p1; such packets must not reach this code.
899 static_assert(kSize >= 2 * Group, "zip_in_place needs at least two lane groups per vector");
900 const VectorT tmp = __builtin_shufflevector(p1, p2, zip_index<Group, kSize>(Is, 0)...);
901 p2 = __builtin_shufflevector(p1, p2, zip_index<Group, kSize>(Is, kSize / (2 * Group))...);
902 p1 = tmp;
903}
904
905// Complex.h specializes this for its packet types, which zip whole complex
906// values rather than individual reals.
907template <typename Packet>
908EIGEN_ALWAYS_INLINE void zip_in_place(Packet& p1, Packet& p2) {
909 zip_in_place_impl<1>(p1, p2, vector_indices<Packet>{});
910}
911
912template <typename Packet>
913EIGEN_ALWAYS_INLINE void ptranspose_impl(PacketBlock<Packet, 2>& kernel) {
914 zip_in_place(kernel.packet[0], kernel.packet[1]);
915}
916
917template <typename Packet>
918EIGEN_ALWAYS_INLINE void ptranspose_impl(PacketBlock<Packet, 4>& kernel) {
919 zip_in_place(kernel.packet[0], kernel.packet[2]);
920 zip_in_place(kernel.packet[1], kernel.packet[3]);
921 zip_in_place(kernel.packet[0], kernel.packet[1]);
922 zip_in_place(kernel.packet[2], kernel.packet[3]);
923}
924
925template <typename Packet>
926EIGEN_ALWAYS_INLINE void ptranspose_impl(PacketBlock<Packet, 8>& kernel) {
927 zip_in_place(kernel.packet[0], kernel.packet[4]);
928 zip_in_place(kernel.packet[1], kernel.packet[5]);
929 zip_in_place(kernel.packet[2], kernel.packet[6]);
930 zip_in_place(kernel.packet[3], kernel.packet[7]);
931
932 zip_in_place(kernel.packet[0], kernel.packet[2]);
933 zip_in_place(kernel.packet[1], kernel.packet[3]);
934 zip_in_place(kernel.packet[4], kernel.packet[6]);
935 zip_in_place(kernel.packet[5], kernel.packet[7]);
936
937 zip_in_place(kernel.packet[0], kernel.packet[1]);
938 zip_in_place(kernel.packet[2], kernel.packet[3]);
939 zip_in_place(kernel.packet[4], kernel.packet[5]);
940 zip_in_place(kernel.packet[6], kernel.packet[7]);
941}
942
943template <typename Packet>
944EIGEN_ALWAYS_INLINE void ptranspose_impl(PacketBlock<Packet, 16>& kernel) {
945 EIGEN_UNROLL_LOOP
946 for (int i = 0; i < 4; ++i) {
947 const int m = (1 << i);
948 EIGEN_UNROLL_LOOP
949 for (int j = 0; j < m; ++j) {
950 const int n = (1 << (3 - i));
951 EIGEN_UNROLL_LOOP
952 for (int k = 0; k < n; ++k) {
953 const int idx = 2 * j * n + k;
954 zip_in_place(kernel.packet[idx], kernel.packet[idx + n]);
955 }
956 }
957 }
958}
959
960} // namespace detail
961
962// ptranspose overloads: only emit valid block sizes per vector size.
963// At 16 bytes: float has 4 elems, double has 2 elems.
964// At 32 bytes: float has 8 elems, double has 4 elems.
965// At 64 bytes: float has 16 elems, double has 8 elems.
966
967// All sizes support PacketBlock<PacketXf, 2> and PacketBlock<PacketXf, 4>.
968EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void ptranspose(PacketBlock<PacketXf, 4>& kernel) {
969 detail::ptranspose_impl(kernel);
970}
971EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void ptranspose(PacketBlock<PacketXf, 2>& kernel) {
972 detail::ptranspose_impl(kernel);
973}
974
975// All sizes support PacketBlock<PacketXd, 2>.
976EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void ptranspose(PacketBlock<PacketXd, 2>& kernel) {
977 detail::ptranspose_impl(kernel);
978}
979
980// All sizes support PacketBlock<PacketXi, 2> and PacketBlock<PacketXi, 4>.
981EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void ptranspose(PacketBlock<PacketXi, 4>& kernel) {
982 detail::ptranspose_impl(kernel);
983}
984EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void ptranspose(PacketBlock<PacketXi, 2>& kernel) {
985 detail::ptranspose_impl(kernel);
986}
987
988// All sizes support PacketBlock<PacketXl, 2>.
989EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void ptranspose(PacketBlock<PacketXl, 2>& kernel) {
990 detail::ptranspose_impl(kernel);
991}
992
993#if EIGEN_GENERIC_VECTOR_SIZE_BYTES >= 32
994// 32+ bytes: float has 8+ elems, double has 4+ elems.
995EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void ptranspose(PacketBlock<PacketXf, 8>& kernel) {
996 detail::ptranspose_impl(kernel);
997}
998EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void ptranspose(PacketBlock<PacketXd, 4>& kernel) {
999 detail::ptranspose_impl(kernel);
1000}
1001EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void ptranspose(PacketBlock<PacketXi, 8>& kernel) {
1002 detail::ptranspose_impl(kernel);
1003}
1004EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void ptranspose(PacketBlock<PacketXl, 4>& kernel) {
1005 detail::ptranspose_impl(kernel);
1006}
1007#endif
1008
1009#if EIGEN_GENERIC_VECTOR_SIZE_BYTES >= 64
1010// 64 bytes: float has 16 elems, double has 8 elems.
1011EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void ptranspose(PacketBlock<PacketXf, 16>& kernel) {
1012 detail::ptranspose_impl(kernel);
1013}
1014EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void ptranspose(PacketBlock<PacketXd, 8>& kernel) {
1015 detail::ptranspose_impl(kernel);
1016}
1017EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void ptranspose(PacketBlock<PacketXi, 16>& kernel) {
1018 detail::ptranspose_impl(kernel);
1019}
1020EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void ptranspose(PacketBlock<PacketXl, 8>& kernel) {
1021 detail::ptranspose_impl(kernel);
1022}
1023#endif
1024#endif
1025
1026} // end namespace internal
1027} // end namespace Eigen
1028
1029#endif // EIGEN_PACKET_MATH_CLANG_H