11#ifndef EIGEN_PACKET_MATH_ALTIVEC_H
12#define EIGEN_PACKET_MATH_ALTIVEC_H
15#include "../../InternalHeaderCheck.h"
21#ifndef EIGEN_CACHEFRIENDLY_PRODUCT_THRESHOLD
22#define EIGEN_CACHEFRIENDLY_PRODUCT_THRESHOLD 4
25#ifndef EIGEN_HAS_SINGLE_INSTRUCTION_MADD
26#define EIGEN_HAS_SINGLE_INSTRUCTION_MADD
30#ifndef EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS
31#define EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS 32
34typedef __vector
float Packet4f;
35typedef __vector
int Packet4i;
36typedef __vector
unsigned int Packet4ui;
37typedef __vector __bool
int Packet4bi;
38typedef __vector
short int Packet8s;
39typedef __vector
unsigned short int Packet8us;
40typedef __vector __bool
short Packet8bi;
41typedef __vector
signed char Packet16c;
42typedef __vector
unsigned char Packet16uc;
43typedef eigen_packet_wrapper<__vector unsigned short int, 0> Packet8bf;
47#define EIGEN_DECLARE_CONST_FAST_Packet4f(NAME, X) Packet4f p4f_##NAME = {X, X, X, X}
49#define EIGEN_DECLARE_CONST_FAST_Packet4i(NAME, X) Packet4i p4i_##NAME = vec_splat_s32(X)
51#define EIGEN_DECLARE_CONST_FAST_Packet4ui(NAME, X) Packet4ui p4ui_##NAME = {X, X, X, X}
53#define EIGEN_DECLARE_CONST_FAST_Packet8us(NAME, X) Packet8us p8us_##NAME = {X, X, X, X, X, X, X, X}
55#define EIGEN_DECLARE_CONST_FAST_Packet16uc(NAME, X) \
56 Packet16uc p16uc_##NAME = {X, X, X, X, X, X, X, X, X, X, X, X, X, X, X, X}
58#define EIGEN_DECLARE_CONST_Packet4f(NAME, X) Packet4f p4f_##NAME = pset1<Packet4f>(X)
60#define EIGEN_DECLARE_CONST_Packet4i(NAME, X) Packet4i p4i_##NAME = pset1<Packet4i>(X)
62#define EIGEN_DECLARE_CONST_Packet2d(NAME, X) Packet2d p2d_##NAME = pset1<Packet2d>(X)
64#define EIGEN_DECLARE_CONST_Packet2l(NAME, X) Packet2l p2l_##NAME = pset1<Packet2l>(X)
66#define EIGEN_DECLARE_CONST_Packet4f_FROM_INT(NAME, X) \
67 const Packet4f p4f_##NAME = reinterpret_cast<Packet4f>(pset1<Packet4i>(X))
70#define DST_CTRL(size, count, stride) (((size) << 24) | ((count) << 16) | (stride))
71#define __UNPACK_TYPE__(PACKETNAME) typename unpacket_traits<PACKETNAME>::type
74static EIGEN_DECLARE_CONST_FAST_Packet4f(ZERO, 0);
75static EIGEN_DECLARE_CONST_FAST_Packet4i(ZERO, 0);
76static EIGEN_DECLARE_CONST_FAST_Packet4i(ONE, 1);
77static EIGEN_DECLARE_CONST_FAST_Packet4i(MINUS16, -16);
78static EIGEN_DECLARE_CONST_FAST_Packet4i(MINUS1, -1);
79static EIGEN_DECLARE_CONST_FAST_Packet4ui(SIGN, 0x80000000u);
80static EIGEN_DECLARE_CONST_FAST_Packet4ui(PREV0DOT5, 0x3EFFFFFFu);
81static EIGEN_DECLARE_CONST_FAST_Packet8us(ONE, 1);
82static Packet4f p4f_MZERO =
83 (Packet4f)vec_sl((Packet4ui)p4i_MINUS1, (Packet4ui)p4i_MINUS1);
85static Packet4f p4f_ONE = vec_ctf(p4i_ONE, 0);
88static Packet4f p4f_COUNTDOWN = {0.0, 1.0, 2.0, 3.0};
89static Packet4i p4i_COUNTDOWN = {0, 1, 2, 3};
90static Packet8s p8s_COUNTDOWN = {0, 1, 2, 3, 4, 5, 6, 7};
91static Packet8us p8us_COUNTDOWN = {0, 1, 2, 3, 4, 5, 6, 7};
93static Packet16c p16c_COUNTDOWN = {0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15};
94static Packet16uc p16uc_COUNTDOWN = {0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15};
96static Packet16uc p16uc_REVERSE32 = {12, 13, 14, 15, 8, 9, 10, 11, 4, 5, 6, 7, 0, 1, 2, 3};
97static Packet16uc p16uc_REVERSE16 = {14, 15, 12, 13, 10, 11, 8, 9, 6, 7, 4, 5, 2, 3, 0, 1};
98static Packet16uc p16uc_REVERSE8 = {15, 14, 13, 12, 11, 10, 9, 8, 7, 6, 5, 4, 3, 2, 1, 0};
101static Packet16uc p16uc_DUPLICATE32_HI = {0, 1, 2, 3, 0, 1, 2, 3, 4, 5, 6, 7, 4, 5, 6, 7};
103static const Packet16uc p16uc_DUPLICATE16_EVEN = {0, 1, 0, 1, 4, 5, 4, 5, 8, 9, 8, 9, 12, 13, 12, 13};
104static const Packet16uc p16uc_DUPLICATE16_ODD = {2, 3, 2, 3, 6, 7, 6, 7, 10, 11, 10, 11, 14, 15, 14, 15};
106static Packet16uc p16uc_QUADRUPLICATE16_HI = {0, 1, 0, 1, 0, 1, 0, 1, 2, 3, 2, 3, 2, 3, 2, 3};
107static Packet16uc p16uc_QUADRUPLICATE16 = {0, 0, 0, 0, 1, 1, 1, 1, 2, 2, 2, 2, 3, 3, 3, 3};
109static Packet16uc p16uc_MERGEE16 = {0, 1, 16, 17, 4, 5, 20, 21, 8, 9, 24, 25, 12, 13, 28, 29};
110static Packet16uc p16uc_MERGEO16 = {2, 3, 18, 19, 6, 7, 22, 23, 10, 11, 26, 27, 14, 15, 30, 31};
112static Packet16uc p16uc_MERGEH16 = {0, 1, 4, 5, 8, 9, 12, 13, 16, 17, 20, 21, 24, 25, 28, 29};
114static Packet16uc p16uc_MERGEL16 = {2, 3, 6, 7, 10, 11, 14, 15, 18, 19, 22, 23, 26, 27, 30, 31};
120static Packet16uc p16uc_FORWARD = vec_lvsl(0, (
float*)0);
121static Packet16uc p16uc_PSET32_WODD =
122 vec_sld((Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 0), (Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 2),
124static Packet16uc p16uc_PSET32_WEVEN = vec_sld(p16uc_DUPLICATE32_HI, (Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 3),
126static Packet16uc p16uc_HALF64_0_16 = vec_sld((Packet16uc)p4i_ZERO, vec_splat((Packet16uc)vec_abs(p4i_MINUS16), 3),
129static Packet16uc p16uc_FORWARD = p16uc_REVERSE32;
130static Packet16uc p16uc_PSET32_WODD =
131 vec_sld((Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 1), (Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 3),
133static Packet16uc p16uc_PSET32_WEVEN =
134 vec_sld((Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 0), (Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 2),
136static Packet16uc p16uc_HALF64_0_16 = vec_sld(vec_splat((Packet16uc)vec_abs(p4i_MINUS16), 0), (Packet16uc)p4i_ZERO,
140static Packet16uc p16uc_PSET64_HI = (Packet16uc)vec_mergeh(
141 (Packet4ui)p16uc_PSET32_WODD, (Packet4ui)p16uc_PSET32_WEVEN);
142static Packet16uc p16uc_PSET64_LO = (Packet16uc)vec_mergel(
143 (Packet4ui)p16uc_PSET32_WODD, (Packet4ui)p16uc_PSET32_WEVEN);
144static Packet16uc p16uc_TRANSPOSE64_HI =
145 p16uc_PSET64_HI + p16uc_HALF64_0_16;
146static Packet16uc p16uc_TRANSPOSE64_LO =
147 p16uc_PSET64_LO + p16uc_HALF64_0_16;
149static Packet16uc p16uc_COMPLEX32_REV =
150 vec_sld(p16uc_REVERSE32, p16uc_REVERSE32, 8);
152#if EIGEN_HAS_BUILTIN(__builtin_prefetch) || EIGEN_COMP_GNUC
153#define EIGEN_PPC_PREFETCH(ADDR) __builtin_prefetch(ADDR);
155#define EIGEN_PPC_PREFETCH(ADDR) asm(" dcbt [%[addr]]\n" ::[addr] "r"(ADDR) : "cc");
159#define LOAD_STORE_UNROLL_16 _Pragma("unroll 16")
161#define LOAD_STORE_UNROLL_16 _Pragma("GCC unroll(16)")
165struct packet_traits<float> : default_packet_traits {
166 typedef Packet4f type;
167 typedef Packet4f half;
180 HasSin = EIGEN_FAST_MATH,
181 HasCos = EIGEN_FAST_MATH,
182 HasTan = EIGEN_FAST_MATH,
191#ifdef EIGEN_VECTORIZE_VSX
201 HasTanh = EIGEN_FAST_MATH,
202 HasErf = EIGEN_FAST_MATH,
203 HasErfc = EIGEN_FAST_MATH,
214struct packet_traits<bfloat16> : default_packet_traits {
215 typedef Packet8bf type;
216 typedef Packet8bf half;
229 HasSin = EIGEN_FAST_MATH,
230 HasCos = EIGEN_FAST_MATH,
233#ifdef EIGEN_VECTORIZE_VSX
251struct packet_traits<int> : default_packet_traits {
252 typedef Packet4i type;
253 typedef Packet4i half;
263#if defined(_ARCH_PWR10) && (EIGEN_COMP_LLVM || EIGEN_GNUC_STRICT_AT_LEAST(11, 0, 0))
273struct packet_traits<short int> : default_packet_traits {
274 typedef Packet8s type;
275 typedef Packet8s half;
290struct packet_traits<unsigned short int> : default_packet_traits {
291 typedef Packet8us type;
292 typedef Packet8us half;
307struct packet_traits<signed char> : default_packet_traits {
308 typedef Packet16c type;
309 typedef Packet16c half;
324struct packet_traits<unsigned char> : default_packet_traits {
325 typedef Packet16uc type;
326 typedef Packet16uc half;
341struct unpacket_traits<Packet4f> {
343 typedef Packet4f half;
344 typedef Packet4i integer_packet;
349 masked_load_available =
false,
350 masked_store_available =
false
354struct unpacket_traits<Packet4i> {
356 typedef Packet4i half;
361 masked_load_available =
false,
362 masked_store_available =
false
366struct unpacket_traits<Packet8s> {
367 typedef short int type;
368 typedef Packet8s half;
373 masked_load_available =
false,
374 masked_store_available =
false
378struct unpacket_traits<Packet8us> {
379 typedef unsigned short int type;
380 typedef Packet8us half;
385 masked_load_available =
false,
386 masked_store_available =
false
391struct unpacket_traits<Packet16c> {
392 typedef signed char type;
393 typedef Packet16c half;
398 masked_load_available =
false,
399 masked_store_available =
false
403struct unpacket_traits<Packet16uc> {
404 typedef unsigned char type;
405 typedef Packet16uc half;
410 masked_load_available =
false,
411 masked_store_available =
false
416struct unpacket_traits<Packet8bf> {
417 typedef bfloat16 type;
418 typedef Packet8bf half;
423 masked_load_available =
false,
424 masked_store_available =
false
428template <
typename Packet>
429EIGEN_STRONG_INLINE Packet pload_common(
const __UNPACK_TYPE__(Packet) * from) {
432 EIGEN_UNUSED_VARIABLE(from);
433 EIGEN_DEBUG_ALIGNED_LOAD
434#ifdef EIGEN_VECTORIZE_VSX
435 return vec_xl(0,
const_cast<__UNPACK_TYPE__(Packet)*
>(from));
437 return vec_ld(0, from);
443EIGEN_STRONG_INLINE Packet4f pload<Packet4f>(
const float* from) {
444 return pload_common<Packet4f>(from);
448EIGEN_STRONG_INLINE Packet4i pload<Packet4i>(
const int* from) {
449 return pload_common<Packet4i>(from);
453EIGEN_STRONG_INLINE Packet8s pload<Packet8s>(
const short int* from) {
454 return pload_common<Packet8s>(from);
458EIGEN_STRONG_INLINE Packet8us pload<Packet8us>(
const unsigned short int* from) {
459 return pload_common<Packet8us>(from);
463EIGEN_STRONG_INLINE Packet16c pload<Packet16c>(
const signed char* from) {
464 return pload_common<Packet16c>(from);
468EIGEN_STRONG_INLINE Packet16uc pload<Packet16uc>(
const unsigned char* from) {
469 return pload_common<Packet16uc>(from);
473EIGEN_STRONG_INLINE Packet8bf pload<Packet8bf>(
const bfloat16* from) {
474 return pload_common<Packet8us>(
reinterpret_cast<const unsigned short int*
>(from));
477template <
typename Packet>
478EIGEN_ALWAYS_INLINE Packet pload_ignore(
const __UNPACK_TYPE__(Packet) * from) {
481 EIGEN_UNUSED_VARIABLE(from);
482 EIGEN_DEBUG_ALIGNED_LOAD
485#pragma GCC diagnostic push
486#pragma GCC diagnostic ignored "-Wmaybe-uninitialized"
488#ifdef EIGEN_VECTORIZE_VSX
489 return vec_xl(0,
const_cast<__UNPACK_TYPE__(Packet)*
>(from));
491 return vec_ld(0, from);
494#pragma GCC diagnostic pop
499EIGEN_ALWAYS_INLINE Packet8bf pload_ignore<Packet8bf>(
const bfloat16* from) {
500 return pload_ignore<Packet8us>(
reinterpret_cast<const unsigned short int*
>(from));
503template <
typename Packet>
504EIGEN_ALWAYS_INLINE Packet pload_partial_common(
const __UNPACK_TYPE__(Packet) * from,
const Index n,
505 const Index offset) {
508 const Index packet_size = unpacket_traits<Packet>::size;
509 eigen_internal_assert(n + offset <= packet_size &&
"number of elements plus offset will read past end of packet");
510 const Index size =
sizeof(__UNPACK_TYPE__(Packet));
512 EIGEN_UNUSED_VARIABLE(packet_size);
513 EIGEN_DEBUG_ALIGNED_LOAD
514 EIGEN_UNUSED_VARIABLE(from);
515 Packet load = vec_xl_len(
const_cast<__UNPACK_TYPE__(Packet)*
>(from), n * size);
517 Packet16uc shift = pset1<Packet16uc>(offset * 8 * size);
519 load = Packet(vec_sro(Packet16uc(load), shift));
521 load = Packet(vec_slo(Packet16uc(load), shift));
527 EIGEN_ALIGN16 __UNPACK_TYPE__(Packet) load[packet_size];
528 unsigned char* load2 =
reinterpret_cast<unsigned char*
>(load + offset);
529 unsigned char* from2 =
reinterpret_cast<unsigned char*
>(
const_cast<__UNPACK_TYPE__(Packet)*
>(from));
532 pstoreu(load2, ploadu<Packet16uc>(from2));
534 memcpy((
void*)load2, (
void*)from2, n2);
536 return pload_ignore<Packet>(load);
538 return Packet(pset1<Packet16uc>(0));
544EIGEN_ALWAYS_INLINE Packet4f pload_partial<Packet4f>(
const float* from,
const Index n,
const Index offset) {
545 return pload_partial_common<Packet4f>(from, n, offset);
549EIGEN_ALWAYS_INLINE Packet4i pload_partial<Packet4i>(
const int* from,
const Index n,
const Index offset) {
550 return pload_partial_common<Packet4i>(from, n, offset);
554EIGEN_ALWAYS_INLINE Packet8s pload_partial<Packet8s>(
const short int* from,
const Index n,
const Index offset) {
555 return pload_partial_common<Packet8s>(from, n, offset);
559EIGEN_ALWAYS_INLINE Packet8us pload_partial<Packet8us>(
const unsigned short int* from,
const Index n,
560 const Index offset) {
561 return pload_partial_common<Packet8us>(from, n, offset);
565EIGEN_ALWAYS_INLINE Packet8bf pload_partial<Packet8bf>(
const bfloat16* from,
const Index n,
const Index offset) {
566 return pload_partial_common<Packet8us>(
reinterpret_cast<const unsigned short int*
>(from), n, offset);
570EIGEN_ALWAYS_INLINE Packet16c pload_partial<Packet16c>(
const signed char* from,
const Index n,
const Index offset) {
571 return pload_partial_common<Packet16c>(from, n, offset);
575EIGEN_ALWAYS_INLINE Packet16uc pload_partial<Packet16uc>(
const unsigned char* from,
const Index n,
const Index offset) {
576 return pload_partial_common<Packet16uc>(from, n, offset);
579template <
typename Packet>
580EIGEN_STRONG_INLINE
void pstore_common(__UNPACK_TYPE__(Packet) * to,
const Packet& from) {
583 EIGEN_UNUSED_VARIABLE(to);
584 EIGEN_DEBUG_ALIGNED_STORE
585#ifdef EIGEN_VECTORIZE_VSX
586 vec_xst(from, 0, to);
593EIGEN_STRONG_INLINE
void pstore<float>(
float* to,
const Packet4f& from) {
594 pstore_common<Packet4f>(to, from);
598EIGEN_STRONG_INLINE
void pstore<int>(
int* to,
const Packet4i& from) {
599 pstore_common<Packet4i>(to, from);
603EIGEN_STRONG_INLINE
void pstore<short int>(
short int* to,
const Packet8s& from) {
604 pstore_common<Packet8s>(to, from);
608EIGEN_STRONG_INLINE
void pstore<unsigned short int>(
unsigned short int* to,
const Packet8us& from) {
609 pstore_common<Packet8us>(to, from);
613EIGEN_STRONG_INLINE
void pstore<bfloat16>(bfloat16* to,
const Packet8bf& from) {
614 pstore_common<Packet8us>(
reinterpret_cast<unsigned short int*
>(to), from.m_val);
618EIGEN_STRONG_INLINE
void pstore<signed char>(
signed char* to,
const Packet16c& from) {
619 pstore_common<Packet16c>(to, from);
623EIGEN_STRONG_INLINE
void pstore<unsigned char>(
unsigned char* to,
const Packet16uc& from) {
624 pstore_common<Packet16uc>(to, from);
627template <
typename Packet>
628EIGEN_ALWAYS_INLINE
void pstore_partial_common(__UNPACK_TYPE__(Packet) * to,
const Packet& from,
const Index n,
629 const Index offset) {
632 const Index packet_size = unpacket_traits<Packet>::size;
633 eigen_internal_assert(n + offset <= packet_size &&
"number of elements plus offset will write past end of packet");
634 const Index size =
sizeof(__UNPACK_TYPE__(Packet));
636 EIGEN_UNUSED_VARIABLE(packet_size);
637 EIGEN_UNUSED_VARIABLE(to);
638 EIGEN_DEBUG_ALIGNED_STORE
641 Packet16uc shift = pset1<Packet16uc>(offset * 8 * size);
643 store = Packet(vec_slo(Packet16uc(store), shift));
645 store = Packet(vec_sro(Packet16uc(store), shift));
648 vec_xst_len(store, to, n * size);
651 EIGEN_ALIGN16 __UNPACK_TYPE__(Packet) store[packet_size];
653 unsigned char* store2 =
reinterpret_cast<unsigned char*
>(store + offset);
654 unsigned char* to2 =
reinterpret_cast<unsigned char*
>(to);
657 pstore(to2, ploadu<Packet16uc>(store2));
659 memcpy((
void*)to2, (
void*)store2, n2);
666EIGEN_ALWAYS_INLINE
void pstore_partial<float>(
float* to,
const Packet4f& from,
const Index n,
const Index offset) {
667 pstore_partial_common<Packet4f>(to, from, n, offset);
671EIGEN_ALWAYS_INLINE
void pstore_partial<int>(
int* to,
const Packet4i& from,
const Index n,
const Index offset) {
672 pstore_partial_common<Packet4i>(to, from, n, offset);
676EIGEN_ALWAYS_INLINE
void pstore_partial<short int>(
short int* to,
const Packet8s& from,
const Index n,
677 const Index offset) {
678 pstore_partial_common<Packet8s>(to, from, n, offset);
682EIGEN_ALWAYS_INLINE
void pstore_partial<unsigned short int>(
unsigned short int* to,
const Packet8us& from,
683 const Index n,
const Index offset) {
684 pstore_partial_common<Packet8us>(to, from, n, offset);
688EIGEN_ALWAYS_INLINE
void pstore_partial<bfloat16>(bfloat16* to,
const Packet8bf& from,
const Index n,
689 const Index offset) {
690 pstore_partial_common<Packet8us>(
reinterpret_cast<unsigned short int*
>(to), from.m_val, n, offset);
694EIGEN_ALWAYS_INLINE
void pstore_partial<signed char>(
signed char* to,
const Packet16c& from,
const Index n,
695 const Index offset) {
696 pstore_partial_common<Packet16c>(to, from, n, offset);
700EIGEN_ALWAYS_INLINE
void pstore_partial<unsigned char>(
unsigned char* to,
const Packet16uc& from,
const Index n,
701 const Index offset) {
702 pstore_partial_common<Packet16uc>(to, from, n, offset);
705template <
typename Packet>
706EIGEN_STRONG_INLINE Packet pset1_size4(
const __UNPACK_TYPE__(Packet) & from) {
707 Packet v = {from, from, from, from};
711template <
typename Packet>
712EIGEN_STRONG_INLINE Packet pset1_size8(
const __UNPACK_TYPE__(Packet) & from) {
713 Packet v = {from, from, from, from, from, from, from, from};
717template <
typename Packet>
718EIGEN_STRONG_INLINE Packet pset1_size16(
const __UNPACK_TYPE__(Packet) & from) {
719 Packet v = {from, from, from, from, from, from, from, from, from, from, from, from, from, from, from, from};
724EIGEN_STRONG_INLINE Packet4f pset1<Packet4f>(
const float& from) {
725 return pset1_size4<Packet4f>(from);
729EIGEN_STRONG_INLINE Packet4i pset1<Packet4i>(
const int& from) {
730 return pset1_size4<Packet4i>(from);
734EIGEN_STRONG_INLINE Packet8s pset1<Packet8s>(
const short int& from) {
735 return pset1_size8<Packet8s>(from);
739EIGEN_STRONG_INLINE Packet8us pset1<Packet8us>(
const unsigned short int& from) {
740 return pset1_size8<Packet8us>(from);
744EIGEN_STRONG_INLINE Packet16c pset1<Packet16c>(
const signed char& from) {
745 return pset1_size16<Packet16c>(from);
749EIGEN_STRONG_INLINE Packet16uc pset1<Packet16uc>(
const unsigned char& from) {
750 return pset1_size16<Packet16uc>(from);
754EIGEN_STRONG_INLINE Packet4f pset1frombits<Packet4f>(
unsigned int from) {
755 return reinterpret_cast<Packet4f
>(pset1<Packet4i>(from));
759EIGEN_STRONG_INLINE Packet8bf pset1<Packet8bf>(
const bfloat16& from) {
760 return pset1_size8<Packet8us>(
reinterpret_cast<const unsigned short int&
>(from));
763template <
typename Packet>
764EIGEN_STRONG_INLINE
void pbroadcast4_common(
const __UNPACK_TYPE__(Packet) * a, Packet& a0, Packet& a1, Packet& a2,
766 a3 = pload<Packet>(a);
767 a0 = vec_splat(a3, 0);
768 a1 = vec_splat(a3, 1);
769 a2 = vec_splat(a3, 2);
770 a3 = vec_splat(a3, 3);
774EIGEN_STRONG_INLINE
void pbroadcast4<Packet4f>(
const float* a, Packet4f& a0, Packet4f& a1, Packet4f& a2, Packet4f& a3) {
775 pbroadcast4_common<Packet4f>(a, a0, a1, a2, a3);
778EIGEN_STRONG_INLINE
void pbroadcast4<Packet4i>(
const int* a, Packet4i& a0, Packet4i& a1, Packet4i& a2, Packet4i& a3) {
779 pbroadcast4_common<Packet4i>(a, a0, a1, a2, a3);
782template <
typename Packet>
783EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet pgather_common(
const __UNPACK_TYPE__(Packet) * from, Index stride,
784 const Index n = unpacket_traits<Packet>::size) {
785 EIGEN_ALIGN16 __UNPACK_TYPE__(Packet) a[unpacket_traits<Packet>::size];
786 eigen_internal_assert(n <= unpacket_traits<Packet>::size &&
"number of elements will gather past end of packet");
788 if (n == unpacket_traits<Packet>::size) {
789 return ploadu<Packet>(from);
791 return ploadu_partial<Packet>(from, n);
795 for (Index i = 0; i < n; i++) {
796 a[i] = from[i * stride];
799 return pload_ignore<Packet>(a);
804EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet4f pgather<float, Packet4f>(
const float* from, Index stride) {
805 return pgather_common<Packet4f>(from, stride);
809EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet4i pgather<int, Packet4i>(
const int* from, Index stride) {
810 return pgather_common<Packet4i>(from, stride);
814EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet8s pgather<short int, Packet8s>(
const short int* from, Index stride) {
815 return pgather_common<Packet8s>(from, stride);
819EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet8us pgather<unsigned short int, Packet8us>(
const unsigned short int* from,
821 return pgather_common<Packet8us>(from, stride);
825EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet8bf pgather<bfloat16, Packet8bf>(
const bfloat16* from, Index stride) {
826 return pgather_common<Packet8bf>(from, stride);
830EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet16c pgather<signed char, Packet16c>(
const signed char* from, Index stride) {
831 return pgather_common<Packet16c>(from, stride);
835EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet16uc pgather<unsigned char, Packet16uc>(
const unsigned char* from,
837 return pgather_common<Packet16uc>(from, stride);
841EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet4f pgather_partial<float, Packet4f>(
const float* from, Index stride,
843 return pgather_common<Packet4f>(from, stride, n);
847EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet4i pgather_partial<int, Packet4i>(
const int* from, Index stride,
849 return pgather_common<Packet4i>(from, stride, n);
853EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet8s pgather_partial<short int, Packet8s>(
const short int* from, Index stride,
855 return pgather_common<Packet8s>(from, stride, n);
859EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet8us
860pgather_partial<unsigned short int, Packet8us>(
const unsigned short int* from, Index stride,
const Index n) {
861 return pgather_common<Packet8us>(from, stride, n);
865EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet8bf pgather_partial<bfloat16, Packet8bf>(
const bfloat16* from, Index stride,
867 return pgather_common<Packet8bf>(from, stride, n);
871EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet16c pgather_partial<signed char, Packet16c>(
const signed char* from,
872 Index stride,
const Index n) {
873 return pgather_common<Packet16c>(from, stride, n);
877EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet16uc pgather_partial<unsigned char, Packet16uc>(
const unsigned char* from,
880 return pgather_common<Packet16uc>(from, stride, n);
883template <
typename Packet>
884EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE
void pscatter_common(__UNPACK_TYPE__(Packet) * to,
const Packet& from,
886 const Index n = unpacket_traits<Packet>::size) {
887 EIGEN_ALIGN16 __UNPACK_TYPE__(Packet) a[unpacket_traits<Packet>::size];
888 eigen_internal_assert(n <= unpacket_traits<Packet>::size &&
"number of elements will scatter past end of packet");
890 if (n == unpacket_traits<Packet>::size) {
891 return pstoreu(to, from);
893 return pstoreu_partial(to, from, n);
896 pstore<__UNPACK_TYPE__(Packet)>(a, from);
898 for (Index i = 0; i < n; i++) {
899 to[i * stride] = a[i];
905EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE
void pscatter<float, Packet4f>(
float* to,
const Packet4f& from, Index stride) {
906 pscatter_common<Packet4f>(to, from, stride);
910EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE
void pscatter<int, Packet4i>(
int* to,
const Packet4i& from, Index stride) {
911 pscatter_common<Packet4i>(to, from, stride);
915EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE
void pscatter<short int, Packet8s>(
short int* to,
const Packet8s& from,
917 pscatter_common<Packet8s>(to, from, stride);
921EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE
void pscatter<unsigned short int, Packet8us>(
unsigned short int* to,
922 const Packet8us& from,
924 pscatter_common<Packet8us>(to, from, stride);
928EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE
void pscatter<bfloat16, Packet8bf>(bfloat16* to,
const Packet8bf& from,
930 pscatter_common<Packet8bf>(to, from, stride);
934EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE
void pscatter<signed char, Packet16c>(
signed char* to,
const Packet16c& from,
936 pscatter_common<Packet16c>(to, from, stride);
940EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE
void pscatter<unsigned char, Packet16uc>(
unsigned char* to,
941 const Packet16uc& from, Index stride) {
942 pscatter_common<Packet16uc>(to, from, stride);
946EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE
void pscatter_partial<float, Packet4f>(
float* to,
const Packet4f& from,
947 Index stride,
const Index n) {
948 pscatter_common<Packet4f>(to, from, stride, n);
952EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE
void pscatter_partial<int, Packet4i>(
int* to,
const Packet4i& from, Index stride,
954 pscatter_common<Packet4i>(to, from, stride, n);
958EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE
void pscatter_partial<short int, Packet8s>(
short int* to,
const Packet8s& from,
959 Index stride,
const Index n) {
960 pscatter_common<Packet8s>(to, from, stride, n);
964EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE
void pscatter_partial<unsigned short int, Packet8us>(
unsigned short int* to,
965 const Packet8us& from,
968 pscatter_common<Packet8us>(to, from, stride, n);
972EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE
void pscatter_partial<bfloat16, Packet8bf>(bfloat16* to,
const Packet8bf& from,
973 Index stride,
const Index n) {
974 pscatter_common<Packet8bf>(to, from, stride, n);
978EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE
void pscatter_partial<signed char, Packet16c>(
signed char* to,
979 const Packet16c& from, Index stride,
981 pscatter_common<Packet16c>(to, from, stride, n);
985EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE
void pscatter_partial<unsigned char, Packet16uc>(
unsigned char* to,
986 const Packet16uc& from,
987 Index stride,
const Index n) {
988 pscatter_common<Packet16uc>(to, from, stride, n);
992EIGEN_STRONG_INLINE Packet4f plset<Packet4f>(
const float& a) {
993 return pset1<Packet4f>(a) + p4f_COUNTDOWN;
996EIGEN_STRONG_INLINE Packet4i plset<Packet4i>(
const int& a) {
997 return pset1<Packet4i>(a) + p4i_COUNTDOWN;
1000EIGEN_STRONG_INLINE Packet8s plset<Packet8s>(
const short int& a) {
1001 return pset1<Packet8s>(a) + p8s_COUNTDOWN;
1004EIGEN_STRONG_INLINE Packet8us plset<Packet8us>(
const unsigned short int& a) {
1005 return pset1<Packet8us>(a) + p8us_COUNTDOWN;
1008EIGEN_STRONG_INLINE Packet16c plset<Packet16c>(
const signed char& a) {
1009 return pset1<Packet16c>(a) + p16c_COUNTDOWN;
1012EIGEN_STRONG_INLINE Packet16uc plset<Packet16uc>(
const unsigned char& a) {
1013 return pset1<Packet16uc>(a) + p16uc_COUNTDOWN;
1017EIGEN_STRONG_INLINE Packet4f padd<Packet4f>(
const Packet4f& a,
const Packet4f& b) {
1021EIGEN_STRONG_INLINE Packet4i padd<Packet4i>(
const Packet4i& a,
const Packet4i& b) {
1025EIGEN_STRONG_INLINE Packet4ui padd<Packet4ui>(
const Packet4ui& a,
const Packet4ui& b) {
1029EIGEN_STRONG_INLINE Packet8s padd<Packet8s>(
const Packet8s& a,
const Packet8s& b) {
1033EIGEN_STRONG_INLINE Packet8us padd<Packet8us>(
const Packet8us& a,
const Packet8us& b) {
1037EIGEN_STRONG_INLINE Packet16c padd<Packet16c>(
const Packet16c& a,
const Packet16c& b) {
1041EIGEN_STRONG_INLINE Packet16uc padd<Packet16uc>(
const Packet16uc& a,
const Packet16uc& b) {
1046EIGEN_STRONG_INLINE Packet4f psub<Packet4f>(
const Packet4f& a,
const Packet4f& b) {
1050EIGEN_STRONG_INLINE Packet4i psub<Packet4i>(
const Packet4i& a,
const Packet4i& b) {
1054EIGEN_STRONG_INLINE Packet8s psub<Packet8s>(
const Packet8s& a,
const Packet8s& b) {
1058EIGEN_STRONG_INLINE Packet8us psub<Packet8us>(
const Packet8us& a,
const Packet8us& b) {
1062EIGEN_STRONG_INLINE Packet16c psub<Packet16c>(
const Packet16c& a,
const Packet16c& b) {
1066EIGEN_STRONG_INLINE Packet16uc psub<Packet16uc>(
const Packet16uc& a,
const Packet16uc& b) {
1071EIGEN_STRONG_INLINE Packet4f pnegate(
const Packet4f& a) {
1072#ifdef EIGEN_VECTORIZE_POWER8_VECTOR
1075 return vec_xor(a, p4f_MZERO);
1079EIGEN_STRONG_INLINE Packet16c pnegate(
const Packet16c& a) {
1080#ifdef EIGEN_VECTORIZE_POWER8_VECTOR
1083 return reinterpret_cast<Packet16c
>(p4i_ZERO) - a;
1087EIGEN_STRONG_INLINE Packet8s pnegate(
const Packet8s& a) {
1088#ifdef EIGEN_VECTORIZE_POWER8_VECTOR
1091 return reinterpret_cast<Packet8s
>(p4i_ZERO) - a;
1095EIGEN_STRONG_INLINE Packet4i pnegate(
const Packet4i& a) {
1096#ifdef EIGEN_VECTORIZE_POWER8_VECTOR
1099 return p4i_ZERO - a;
1104EIGEN_STRONG_INLINE Packet4f pmul<Packet4f>(
const Packet4f& a,
const Packet4f& b) {
1105 return vec_madd(a, b, p4f_MZERO);
1108EIGEN_STRONG_INLINE Packet4i pmul<Packet4i>(
const Packet4i& a,
const Packet4i& b) {
1112EIGEN_STRONG_INLINE Packet8s pmul<Packet8s>(
const Packet8s& a,
const Packet8s& b) {
1113 return vec_mul(a, b);
1116EIGEN_STRONG_INLINE Packet8us pmul<Packet8us>(
const Packet8us& a,
const Packet8us& b) {
1117 return vec_mul(a, b);
1120EIGEN_STRONG_INLINE Packet16c pmul<Packet16c>(
const Packet16c& a,
const Packet16c& b) {
1121 return vec_mul(a, b);
1124EIGEN_STRONG_INLINE Packet16uc pmul<Packet16uc>(
const Packet16uc& a,
const Packet16uc& b) {
1125 return vec_mul(a, b);
1129EIGEN_STRONG_INLINE Packet4f pdiv<Packet4f>(
const Packet4f& a,
const Packet4f& b) {
1131 Packet4f t, y_0, y_1;
1137 t = vec_nmsub(y_0, b, p4f_ONE);
1138 y_1 = vec_madd(y_0, t, y_0);
1140 return vec_madd(a, y_1, p4f_MZERO);
1142 return vec_div(a, b);
1147EIGEN_STRONG_INLINE Packet4i pdiv<Packet4i>(
const Packet4i& a,
const Packet4i& b) {
1148#if defined(_ARCH_PWR10) && (EIGEN_COMP_LLVM || EIGEN_GNUC_STRICT_AT_LEAST(11, 0, 0))
1149 return vec_div(a, b);
1151 EIGEN_UNUSED_VARIABLE(a);
1152 EIGEN_UNUSED_VARIABLE(b);
1153 eigen_assert(
false &&
"packet integer division are not supported by AltiVec");
1154 return pset1<Packet4i>(0);
1160EIGEN_STRONG_INLINE Packet4f pmadd(
const Packet4f& a,
const Packet4f& b,
const Packet4f& c) {
1161 return vec_madd(a, b, c);
1164EIGEN_STRONG_INLINE Packet4i pmadd(
const Packet4i& a,
const Packet4i& b,
const Packet4i& c) {
1168EIGEN_STRONG_INLINE Packet8s pmadd(
const Packet8s& a,
const Packet8s& b,
const Packet8s& c) {
1169 return vec_madd(a, b, c);
1172EIGEN_STRONG_INLINE Packet8us pmadd(
const Packet8us& a,
const Packet8us& b,
const Packet8us& c) {
1173 return vec_madd(a, b, c);
1177EIGEN_STRONG_INLINE Packet4f pmsub(
const Packet4f& a,
const Packet4f& b,
const Packet4f& c) {
1178#ifdef EIGEN_VECTORIZE_VSX
1179 return vec_msub(a, b, c);
1181 return vec_madd(a, b, pnegate(c));
1185#ifdef EIGEN_VECTORIZE_VSX
1187EIGEN_STRONG_INLINE Packet4f pnmadd(
const Packet4f& a,
const Packet4f& b,
const Packet4f& c) {
1188 return vec_nmsub(a, b, c);
1191EIGEN_STRONG_INLINE Packet4f pnmsub(
const Packet4f& a,
const Packet4f& b,
const Packet4f& c) {
1192 return vec_nmadd(a, b, c);
1197EIGEN_STRONG_INLINE Packet4f pmin<Packet4f>(
const Packet4f& a,
const Packet4f& b) {
1198#ifdef EIGEN_VECTORIZE_VSX
1201 __asm__(
"xvcmpgesp %x0,%x1,%x2\n\txxsel %x0,%x1,%x2,%x0" :
"=&wa"(ret) :
"wa"(a),
"wa"(b));
1204 return vec_min(a, b);
1208EIGEN_STRONG_INLINE Packet4i pmin<Packet4i>(
const Packet4i& a,
const Packet4i& b) {
1209 return vec_min(a, b);
1212EIGEN_STRONG_INLINE Packet8s pmin<Packet8s>(
const Packet8s& a,
const Packet8s& b) {
1213 return vec_min(a, b);
1216EIGEN_STRONG_INLINE Packet8us pmin<Packet8us>(
const Packet8us& a,
const Packet8us& b) {
1217 return vec_min(a, b);
1220EIGEN_STRONG_INLINE Packet16c pmin<Packet16c>(
const Packet16c& a,
const Packet16c& b) {
1221 return vec_min(a, b);
1224EIGEN_STRONG_INLINE Packet16uc pmin<Packet16uc>(
const Packet16uc& a,
const Packet16uc& b) {
1225 return vec_min(a, b);
1229EIGEN_STRONG_INLINE Packet4f pmax<Packet4f>(
const Packet4f& a,
const Packet4f& b) {
1230#ifdef EIGEN_VECTORIZE_VSX
1233 __asm__(
"xvcmpgtsp %x0,%x2,%x1\n\txxsel %x0,%x1,%x2,%x0" :
"=&wa"(ret) :
"wa"(a),
"wa"(b));
1236 return vec_max(a, b);
1240EIGEN_STRONG_INLINE Packet4i pmax<Packet4i>(
const Packet4i& a,
const Packet4i& b) {
1241 return vec_max(a, b);
1244EIGEN_STRONG_INLINE Packet8s pmax<Packet8s>(
const Packet8s& a,
const Packet8s& b) {
1245 return vec_max(a, b);
1248EIGEN_STRONG_INLINE Packet8us pmax<Packet8us>(
const Packet8us& a,
const Packet8us& b) {
1249 return vec_max(a, b);
1252EIGEN_STRONG_INLINE Packet16c pmax<Packet16c>(
const Packet16c& a,
const Packet16c& b) {
1253 return vec_max(a, b);
1256EIGEN_STRONG_INLINE Packet16uc pmax<Packet16uc>(
const Packet16uc& a,
const Packet16uc& b) {
1257 return vec_max(a, b);
1261EIGEN_STRONG_INLINE Packet4f pcmp_le(
const Packet4f& a,
const Packet4f& b) {
1262 return reinterpret_cast<Packet4f
>(vec_cmple(a, b));
1265#ifdef EIGEN_VECTORIZE_VSX
1267EIGEN_STRONG_INLINE Packet4f pcmp_lt(
const Packet4f& a,
const Packet4f& b) {
1268 return reinterpret_cast<Packet4f
>(vec_cmplt(a, b));
1272EIGEN_STRONG_INLINE Packet4f pcmp_eq(
const Packet4f& a,
const Packet4f& b) {
1273 return reinterpret_cast<Packet4f
>(vec_cmpeq(a, b));
1276EIGEN_STRONG_INLINE Packet4f pcmp_lt_or_nan(
const Packet4f& a,
const Packet4f& b) {
1277 Packet4f c =
reinterpret_cast<Packet4f
>(vec_cmpge(a, b));
1278 return vec_nor(c, c);
1281#ifdef EIGEN_VECTORIZE_VSX
1283EIGEN_STRONG_INLINE Packet4i pcmp_le(
const Packet4i& a,
const Packet4i& b) {
1284 return reinterpret_cast<Packet4i
>(vec_cmple(a, b));
1288EIGEN_STRONG_INLINE Packet4i pcmp_lt(
const Packet4i& a,
const Packet4i& b) {
1289 return reinterpret_cast<Packet4i
>(vec_cmplt(a, b));
1292EIGEN_STRONG_INLINE Packet4i pcmp_eq(
const Packet4i& a,
const Packet4i& b) {
1293 return reinterpret_cast<Packet4i
>(vec_cmpeq(a, b));
1295#ifdef EIGEN_VECTORIZE_VSX
1297EIGEN_STRONG_INLINE Packet8s pcmp_le(
const Packet8s& a,
const Packet8s& b) {
1298 return reinterpret_cast<Packet8s
>(vec_cmple(a, b));
1302EIGEN_STRONG_INLINE Packet8s pcmp_lt(
const Packet8s& a,
const Packet8s& b) {
1303 return reinterpret_cast<Packet8s
>(vec_cmplt(a, b));
1306EIGEN_STRONG_INLINE Packet8s pcmp_eq(
const Packet8s& a,
const Packet8s& b) {
1307 return reinterpret_cast<Packet8s
>(vec_cmpeq(a, b));
1309#ifdef EIGEN_VECTORIZE_VSX
1311EIGEN_STRONG_INLINE Packet8us pcmp_le(
const Packet8us& a,
const Packet8us& b) {
1312 return reinterpret_cast<Packet8us
>(vec_cmple(a, b));
1316EIGEN_STRONG_INLINE Packet8us pcmp_lt(
const Packet8us& a,
const Packet8us& b) {
1317 return reinterpret_cast<Packet8us
>(vec_cmplt(a, b));
1320EIGEN_STRONG_INLINE Packet8us pcmp_eq(
const Packet8us& a,
const Packet8us& b) {
1321 return reinterpret_cast<Packet8us
>(vec_cmpeq(a, b));
1323#ifdef EIGEN_VECTORIZE_VSX
1325EIGEN_STRONG_INLINE Packet16c pcmp_le(
const Packet16c& a,
const Packet16c& b) {
1326 return reinterpret_cast<Packet16c
>(vec_cmple(a, b));
1330EIGEN_STRONG_INLINE Packet16c pcmp_lt(
const Packet16c& a,
const Packet16c& b) {
1331 return reinterpret_cast<Packet16c
>(vec_cmplt(a, b));
1334EIGEN_STRONG_INLINE Packet16c pcmp_eq(
const Packet16c& a,
const Packet16c& b) {
1335 return reinterpret_cast<Packet16c
>(vec_cmpeq(a, b));
1337#ifdef EIGEN_VECTORIZE_VSX
1339EIGEN_STRONG_INLINE Packet16uc pcmp_le(
const Packet16uc& a,
const Packet16uc& b) {
1340 return reinterpret_cast<Packet16uc
>(vec_cmple(a, b));
1344EIGEN_STRONG_INLINE Packet16uc pcmp_lt(
const Packet16uc& a,
const Packet16uc& b) {
1345 return reinterpret_cast<Packet16uc
>(vec_cmplt(a, b));
1348EIGEN_STRONG_INLINE Packet16uc pcmp_eq(
const Packet16uc& a,
const Packet16uc& b) {
1349 return reinterpret_cast<Packet16uc
>(vec_cmpeq(a, b));
1353EIGEN_STRONG_INLINE Packet4f pand<Packet4f>(
const Packet4f& a,
const Packet4f& b) {
1354 return vec_and(a, b);
1357EIGEN_STRONG_INLINE Packet4i pand<Packet4i>(
const Packet4i& a,
const Packet4i& b) {
1358 return vec_and(a, b);
1361EIGEN_STRONG_INLINE Packet4ui pand<Packet4ui>(
const Packet4ui& a,
const Packet4ui& b) {
1362 return vec_and(a, b);
1365EIGEN_STRONG_INLINE Packet8us pand<Packet8us>(
const Packet8us& a,
const Packet8us& b) {
1366 return vec_and(a, b);
1369EIGEN_STRONG_INLINE Packet8bf pand<Packet8bf>(
const Packet8bf& a,
const Packet8bf& b) {
1370 return pand<Packet8us>(a, b);
1374EIGEN_STRONG_INLINE Packet4f por<Packet4f>(
const Packet4f& a,
const Packet4f& b) {
1375 return vec_or(a, b);
1378EIGEN_STRONG_INLINE Packet4i por<Packet4i>(
const Packet4i& a,
const Packet4i& b) {
1379 return vec_or(a, b);
1382EIGEN_STRONG_INLINE Packet8s por<Packet8s>(
const Packet8s& a,
const Packet8s& b) {
1383 return vec_or(a, b);
1386EIGEN_STRONG_INLINE Packet8us por<Packet8us>(
const Packet8us& a,
const Packet8us& b) {
1387 return vec_or(a, b);
1390EIGEN_STRONG_INLINE Packet8bf por<Packet8bf>(
const Packet8bf& a,
const Packet8bf& b) {
1391 return por<Packet8us>(a, b);
1395EIGEN_STRONG_INLINE Packet4f pxor<Packet4f>(
const Packet4f& a,
const Packet4f& b) {
1396 return vec_xor(a, b);
1399EIGEN_STRONG_INLINE Packet4i pxor<Packet4i>(
const Packet4i& a,
const Packet4i& b) {
1400 return vec_xor(a, b);
1403EIGEN_STRONG_INLINE Packet8us pxor<Packet8us>(
const Packet8us& a,
const Packet8us& b) {
1404 return vec_xor(a, b);
1407EIGEN_STRONG_INLINE Packet8bf pxor<Packet8bf>(
const Packet8bf& a,
const Packet8bf& b) {
1408 return pxor<Packet8us>(a, b);
1412EIGEN_STRONG_INLINE Packet4f pandnot<Packet4f>(
const Packet4f& a,
const Packet4f& b) {
1413 return vec_andc(a, b);
1416EIGEN_STRONG_INLINE Packet4i pandnot<Packet4i>(
const Packet4i& a,
const Packet4i& b) {
1417 return vec_andc(a, b);
1421EIGEN_STRONG_INLINE Packet4f pselect(
const Packet4f& mask,
const Packet4f& a,
const Packet4f& b) {
1422 return vec_sel(b, a,
reinterpret_cast<Packet4ui
>(mask));
1426EIGEN_STRONG_INLINE Packet4f pround<Packet4f>(
const Packet4f& a) {
1427 Packet4f t = vec_add(
1428 reinterpret_cast<Packet4f
>(vec_or(vec_and(
reinterpret_cast<Packet4ui
>(a), p4ui_SIGN), p4ui_PREV0DOT5)), a);
1431#ifdef EIGEN_VECTORIZE_VSX
1432 __asm__(
"xvrspiz %x0, %x1\n\t" :
"=&wa"(res) :
"wa"(t));
1434 __asm__(
"vrfiz %0, %1\n\t" :
"=v"(res) :
"v"(t));
1440EIGEN_STRONG_INLINE Packet4f pceil<Packet4f>(
const Packet4f& a) {
1444EIGEN_STRONG_INLINE Packet4f pfloor<Packet4f>(
const Packet4f& a) {
1445 return vec_floor(a);
1448EIGEN_STRONG_INLINE Packet4f ptrunc<Packet4f>(
const Packet4f& a) {
1449 return vec_trunc(a);
1451#ifdef EIGEN_VECTORIZE_VSX
1453EIGEN_STRONG_INLINE Packet4f print<Packet4f>(
const Packet4f& a) {
1456 __asm__(
"xvrspic %x0, %x1\n\t" :
"=&wa"(res) :
"wa"(a));
1462template <
typename Packet>
1463EIGEN_STRONG_INLINE Packet ploadu_common(
const __UNPACK_TYPE__(Packet) * from) {
1464 EIGEN_DEBUG_UNALIGNED_LOAD
1465#if defined(EIGEN_VECTORIZE_VSX)
1466 return vec_xl(0,
const_cast<__UNPACK_TYPE__(Packet)*
>(from));
1468 Packet16uc MSQ = vec_ld(0, (
unsigned char*)from);
1469 Packet16uc LSQ = vec_ld(15, (
unsigned char*)from);
1470 Packet16uc mask = vec_lvsl(0, from);
1472 return (Packet)vec_perm(MSQ, LSQ, mask);
1477EIGEN_STRONG_INLINE Packet4f ploadu<Packet4f>(
const float* from) {
1478 return ploadu_common<Packet4f>(from);
1481EIGEN_STRONG_INLINE Packet4i ploadu<Packet4i>(
const int* from) {
1482 return ploadu_common<Packet4i>(from);
1485EIGEN_STRONG_INLINE Packet8s ploadu<Packet8s>(
const short int* from) {
1486 return ploadu_common<Packet8s>(from);
1489EIGEN_STRONG_INLINE Packet8us ploadu<Packet8us>(
const unsigned short int* from) {
1490 return ploadu_common<Packet8us>(from);
1493EIGEN_STRONG_INLINE Packet8bf ploadu<Packet8bf>(
const bfloat16* from) {
1494 return ploadu_common<Packet8us>(
reinterpret_cast<const unsigned short int*
>(from));
1497EIGEN_STRONG_INLINE Packet16c ploadu<Packet16c>(
const signed char* from) {
1498 return ploadu_common<Packet16c>(from);
1501EIGEN_STRONG_INLINE Packet16uc ploadu<Packet16uc>(
const unsigned char* from) {
1502 return ploadu_common<Packet16uc>(from);
1505template <
typename Packet>
1506EIGEN_ALWAYS_INLINE Packet ploadu_partial_common(
const __UNPACK_TYPE__(Packet) * from,
const Index n,
1507 const Index offset) {
1508 const Index packet_size = unpacket_traits<Packet>::size;
1509 eigen_internal_assert(n + offset <= packet_size &&
"number of elements plus offset will read past end of packet");
1510 const Index size =
sizeof(__UNPACK_TYPE__(Packet));
1512 EIGEN_UNUSED_VARIABLE(packet_size);
1513 EIGEN_DEBUG_ALIGNED_LOAD
1514 EIGEN_DEBUG_UNALIGNED_LOAD
1515 Packet load = vec_xl_len(
const_cast<__UNPACK_TYPE__(Packet)*
>(from), n * size);
1517 Packet16uc shift = pset1<Packet16uc>(offset * 8 * size);
1519 load = Packet(vec_sro(Packet16uc(load), shift));
1521 load = Packet(vec_slo(Packet16uc(load), shift));
1527 EIGEN_ALIGN16 __UNPACK_TYPE__(Packet) load[packet_size];
1528 unsigned char* load2 =
reinterpret_cast<unsigned char*
>(load + offset);
1529 unsigned char* from2 =
reinterpret_cast<unsigned char*
>(
const_cast<__UNPACK_TYPE__(Packet)*
>(from));
1530 Index n2 = n * size;
1532 pstoreu(load2, ploadu<Packet16uc>(from2));
1534 memcpy((
void*)load2, (
void*)from2, n2);
1536 return pload_ignore<Packet>(load);
1538 return Packet(pset1<Packet16uc>(0));
1544EIGEN_ALWAYS_INLINE Packet4f ploadu_partial<Packet4f>(
const float* from,
const Index n,
const Index offset) {
1545 return ploadu_partial_common<Packet4f>(from, n, offset);
1548EIGEN_ALWAYS_INLINE Packet4i ploadu_partial<Packet4i>(
const int* from,
const Index n,
const Index offset) {
1549 return ploadu_partial_common<Packet4i>(from, n, offset);
1552EIGEN_ALWAYS_INLINE Packet8s ploadu_partial<Packet8s>(
const short int* from,
const Index n,
const Index offset) {
1553 return ploadu_partial_common<Packet8s>(from, n, offset);
1556EIGEN_ALWAYS_INLINE Packet8us ploadu_partial<Packet8us>(
const unsigned short int* from,
const Index n,
1557 const Index offset) {
1558 return ploadu_partial_common<Packet8us>(from, n, offset);
1561EIGEN_ALWAYS_INLINE Packet8bf ploadu_partial<Packet8bf>(
const bfloat16* from,
const Index n,
const Index offset) {
1562 return ploadu_partial_common<Packet8us>(
reinterpret_cast<const unsigned short int*
>(from), n, offset);
1565EIGEN_ALWAYS_INLINE Packet16c ploadu_partial<Packet16c>(
const signed char* from,
const Index n,
const Index offset) {
1566 return ploadu_partial_common<Packet16c>(from, n, offset);
1569EIGEN_ALWAYS_INLINE Packet16uc ploadu_partial<Packet16uc>(
const unsigned char* from,
const Index n,
1570 const Index offset) {
1571 return ploadu_partial_common<Packet16uc>(from, n, offset);
1574template <
typename Packet>
1575EIGEN_STRONG_INLINE Packet ploaddup_common(
const __UNPACK_TYPE__(Packet) * from) {
1577 if ((std::ptrdiff_t(from) % 16) == 0)
1578 p = pload<Packet>(from);
1580 p = ploadu<Packet>(from);
1581 return vec_mergeh(p, p);
1584EIGEN_STRONG_INLINE Packet4f ploaddup<Packet4f>(
const float* from) {
1585 return ploaddup_common<Packet4f>(from);
1588EIGEN_STRONG_INLINE Packet4i ploaddup<Packet4i>(
const int* from) {
1589 return ploaddup_common<Packet4i>(from);
1593EIGEN_STRONG_INLINE Packet8s ploaddup<Packet8s>(
const short int* from) {
1595 if ((std::ptrdiff_t(from) % 16) == 0)
1596 p = pload<Packet8s>(from);
1598 p = ploadu<Packet8s>(from);
1599 return vec_mergeh(p, p);
1603EIGEN_STRONG_INLINE Packet8us ploaddup<Packet8us>(
const unsigned short int* from) {
1605 if ((std::ptrdiff_t(from) % 16) == 0)
1606 p = pload<Packet8us>(from);
1608 p = ploadu<Packet8us>(from);
1609 return vec_mergeh(p, p);
1613EIGEN_STRONG_INLINE Packet8s ploadquad<Packet8s>(
const short int* from) {
1615 if ((std::ptrdiff_t(from) % 16) == 0)
1616 p = pload<Packet8s>(from);
1618 p = ploadu<Packet8s>(from);
1619 return vec_perm(p, p, p16uc_QUADRUPLICATE16_HI);
1623EIGEN_STRONG_INLINE Packet8us ploadquad<Packet8us>(
const unsigned short int* from) {
1625 if ((std::ptrdiff_t(from) % 16) == 0)
1626 p = pload<Packet8us>(from);
1628 p = ploadu<Packet8us>(from);
1629 return vec_perm(p, p, p16uc_QUADRUPLICATE16_HI);
1633EIGEN_STRONG_INLINE Packet8bf ploadquad<Packet8bf>(
const bfloat16* from) {
1634 return ploadquad<Packet8us>(
reinterpret_cast<const unsigned short int*
>(from));
1638EIGEN_STRONG_INLINE Packet16c ploaddup<Packet16c>(
const signed char* from) {
1640 if ((std::ptrdiff_t(from) % 16) == 0)
1641 p = pload<Packet16c>(from);
1643 p = ploadu<Packet16c>(from);
1644 return vec_mergeh(p, p);
1648EIGEN_STRONG_INLINE Packet16uc ploaddup<Packet16uc>(
const unsigned char* from) {
1650 if ((std::ptrdiff_t(from) % 16) == 0)
1651 p = pload<Packet16uc>(from);
1653 p = ploadu<Packet16uc>(from);
1654 return vec_mergeh(p, p);
1658EIGEN_STRONG_INLINE Packet16c ploadquad<Packet16c>(
const signed char* from) {
1660 if ((std::ptrdiff_t(from) % 16) == 0)
1661 p = pload<Packet16c>(from);
1663 p = ploadu<Packet16c>(from);
1664 return vec_perm(p, p, p16uc_QUADRUPLICATE16);
1668EIGEN_STRONG_INLINE Packet16uc ploadquad<Packet16uc>(
const unsigned char* from) {
1670 if ((std::ptrdiff_t(from) % 16) == 0)
1671 p = pload<Packet16uc>(from);
1673 p = ploadu<Packet16uc>(from);
1674 return vec_perm(p, p, p16uc_QUADRUPLICATE16);
1677template <
typename Packet>
1678EIGEN_STRONG_INLINE
void pstoreu_common(__UNPACK_TYPE__(Packet) * to,
const Packet& from) {
1679 EIGEN_DEBUG_UNALIGNED_STORE
1680#if defined(EIGEN_VECTORIZE_VSX)
1681 vec_xst(from, 0, to);
1685 Packet16uc MSQ, LSQ, edges;
1686 Packet16uc edgeAlign, align;
1688 MSQ = vec_ld(0, (
unsigned char*)to);
1689 LSQ = vec_ld(15, (
unsigned char*)to);
1690 edgeAlign = vec_lvsl(0, to);
1691 edges = vec_perm(LSQ, MSQ, edgeAlign);
1692 align = vec_lvsr(0, to);
1693 MSQ = vec_perm(edges, (Packet16uc)from, align);
1694 LSQ = vec_perm((Packet16uc)from, edges, align);
1695 vec_st(LSQ, 15, (
unsigned char*)to);
1696 vec_st(MSQ, 0, (
unsigned char*)to);
1700EIGEN_STRONG_INLINE
void pstoreu<float>(
float* to,
const Packet4f& from) {
1701 pstoreu_common<Packet4f>(to, from);
1704EIGEN_STRONG_INLINE
void pstoreu<int>(
int* to,
const Packet4i& from) {
1705 pstoreu_common<Packet4i>(to, from);
1708EIGEN_STRONG_INLINE
void pstoreu<short int>(
short int* to,
const Packet8s& from) {
1709 pstoreu_common<Packet8s>(to, from);
1712EIGEN_STRONG_INLINE
void pstoreu<unsigned short int>(
unsigned short int* to,
const Packet8us& from) {
1713 pstoreu_common<Packet8us>(to, from);
1716EIGEN_STRONG_INLINE
void pstoreu<bfloat16>(bfloat16* to,
const Packet8bf& from) {
1717 pstoreu_common<Packet8us>(
reinterpret_cast<unsigned short int*
>(to), from.m_val);
1720EIGEN_STRONG_INLINE
void pstoreu<signed char>(
signed char* to,
const Packet16c& from) {
1721 pstoreu_common<Packet16c>(to, from);
1724EIGEN_STRONG_INLINE
void pstoreu<unsigned char>(
unsigned char* to,
const Packet16uc& from) {
1725 pstoreu_common<Packet16uc>(to, from);
1728template <
typename Packet>
1729EIGEN_ALWAYS_INLINE
void pstoreu_partial_common(__UNPACK_TYPE__(Packet) * to,
const Packet& from,
const Index n,
1730 const Index offset) {
1731 const Index packet_size = unpacket_traits<Packet>::size;
1732 eigen_internal_assert(n + offset <= packet_size &&
"number of elements plus offset will write past end of packet");
1733 const Index size =
sizeof(__UNPACK_TYPE__(Packet));
1735 EIGEN_UNUSED_VARIABLE(packet_size);
1736 EIGEN_DEBUG_UNALIGNED_STORE
1737 Packet store = from;
1739 Packet16uc shift = pset1<Packet16uc>(offset * 8 * size);
1741 store = Packet(vec_slo(Packet16uc(store), shift));
1743 store = Packet(vec_sro(Packet16uc(store), shift));
1746 vec_xst_len(store, to, n * size);
1749 EIGEN_ALIGN16 __UNPACK_TYPE__(Packet) store[packet_size];
1750 pstore(store, from);
1751 unsigned char* store2 =
reinterpret_cast<unsigned char*
>(store + offset);
1752 unsigned char* to2 =
reinterpret_cast<unsigned char*
>(to);
1753 Index n2 = n * size;
1755 pstoreu(to2, ploadu<Packet16uc>(store2));
1757 memcpy((
void*)to2, (
void*)store2, n2);
1764EIGEN_ALWAYS_INLINE
void pstoreu_partial<float>(
float* to,
const Packet4f& from,
const Index n,
const Index offset) {
1765 pstoreu_partial_common<Packet4f>(to, from, n, offset);
1768EIGEN_ALWAYS_INLINE
void pstoreu_partial<int>(
int* to,
const Packet4i& from,
const Index n,
const Index offset) {
1769 pstoreu_partial_common<Packet4i>(to, from, n, offset);
1772EIGEN_ALWAYS_INLINE
void pstoreu_partial<short int>(
short int* to,
const Packet8s& from,
const Index n,
1773 const Index offset) {
1774 pstoreu_partial_common<Packet8s>(to, from, n, offset);
1777EIGEN_ALWAYS_INLINE
void pstoreu_partial<unsigned short int>(
unsigned short int* to,
const Packet8us& from,
1778 const Index n,
const Index offset) {
1779 pstoreu_partial_common<Packet8us>(to, from, n, offset);
1782EIGEN_ALWAYS_INLINE
void pstoreu_partial<bfloat16>(bfloat16* to,
const Packet8bf& from,
const Index n,
1783 const Index offset) {
1784 pstoreu_partial_common<Packet8us>(
reinterpret_cast<unsigned short int*
>(to), from, n, offset);
1787EIGEN_ALWAYS_INLINE
void pstoreu_partial<signed char>(
signed char* to,
const Packet16c& from,
const Index n,
1788 const Index offset) {
1789 pstoreu_partial_common<Packet16c>(to, from, n, offset);
1792EIGEN_ALWAYS_INLINE
void pstoreu_partial<unsigned char>(
unsigned char* to,
const Packet16uc& from,
const Index n,
1793 const Index offset) {
1794 pstoreu_partial_common<Packet16uc>(to, from, n, offset);
1798EIGEN_STRONG_INLINE
void prefetch<float>(
const float* addr) {
1799 EIGEN_PPC_PREFETCH(addr);
1802EIGEN_STRONG_INLINE
void prefetch<int>(
const int* addr) {
1803 EIGEN_PPC_PREFETCH(addr);
1807EIGEN_STRONG_INLINE
float pfirst<Packet4f>(
const Packet4f& a) {
1808 EIGEN_ALIGN16
float x;
1813EIGEN_STRONG_INLINE
int pfirst<Packet4i>(
const Packet4i& a) {
1814 EIGEN_ALIGN16
int x;
1819template <
typename Packet>
1820EIGEN_STRONG_INLINE __UNPACK_TYPE__(Packet) pfirst_common(
const Packet& a) {
1821 EIGEN_ALIGN16 __UNPACK_TYPE__(Packet) x;
1827EIGEN_STRONG_INLINE
short int pfirst<Packet8s>(
const Packet8s& a) {
1828 return pfirst_common<Packet8s>(a);
1832EIGEN_STRONG_INLINE
unsigned short int pfirst<Packet8us>(
const Packet8us& a) {
1833 return pfirst_common<Packet8us>(a);
1837EIGEN_STRONG_INLINE
signed char pfirst<Packet16c>(
const Packet16c& a) {
1838 return pfirst_common<Packet16c>(a);
1842EIGEN_STRONG_INLINE
unsigned char pfirst<Packet16uc>(
const Packet16uc& a) {
1843 return pfirst_common<Packet16uc>(a);
1847EIGEN_STRONG_INLINE Packet4f preverse(
const Packet4f& a) {
1848 return reinterpret_cast<Packet4f
>(
1849 vec_perm(
reinterpret_cast<Packet16uc
>(a),
reinterpret_cast<Packet16uc
>(a), p16uc_REVERSE32));
1852EIGEN_STRONG_INLINE Packet4i preverse(
const Packet4i& a) {
1853 return reinterpret_cast<Packet4i
>(
1854 vec_perm(
reinterpret_cast<Packet16uc
>(a),
reinterpret_cast<Packet16uc
>(a), p16uc_REVERSE32));
1857EIGEN_STRONG_INLINE Packet8s preverse(
const Packet8s& a) {
1858 return reinterpret_cast<Packet8s
>(
1859 vec_perm(
reinterpret_cast<Packet16uc
>(a),
reinterpret_cast<Packet16uc
>(a), p16uc_REVERSE16));
1862EIGEN_STRONG_INLINE Packet8us preverse(
const Packet8us& a) {
1863 return reinterpret_cast<Packet8us
>(
1864 vec_perm(
reinterpret_cast<Packet16uc
>(a),
reinterpret_cast<Packet16uc
>(a), p16uc_REVERSE16));
1867EIGEN_STRONG_INLINE Packet16c preverse(
const Packet16c& a) {
1868 return vec_perm(a, a, p16uc_REVERSE8);
1871EIGEN_STRONG_INLINE Packet16uc preverse(
const Packet16uc& a) {
1872 return vec_perm(a, a, p16uc_REVERSE8);
1875EIGEN_STRONG_INLINE Packet8bf preverse(
const Packet8bf& a) {
1876 return preverse<Packet8us>(a);
1880EIGEN_STRONG_INLINE Packet4f pabs(
const Packet4f& a) {
1884EIGEN_STRONG_INLINE Packet4i pabs(
const Packet4i& a) {
1888EIGEN_STRONG_INLINE Packet8s pabs(
const Packet8s& a) {
1892EIGEN_STRONG_INLINE Packet16c pabs(
const Packet16c& a) {
1896EIGEN_STRONG_INLINE Packet8bf pabs(
const Packet8bf& a) {
1897 EIGEN_DECLARE_CONST_FAST_Packet8us(abs_mask, 0x7FFF);
1898 return pand<Packet8us>(p8us_abs_mask, a);
1902EIGEN_STRONG_INLINE Packet8bf psignbit(
const Packet8bf& a) {
1903 return vec_sra(a.m_val, vec_splat_u16(15));
1906EIGEN_STRONG_INLINE Packet4f psignbit(
const Packet4f& a) {
1907 return (Packet4f)vec_sra((Packet4i)a, vec_splats((
unsigned int)(31)));
1911EIGEN_STRONG_INLINE Packet4i parithmetic_shift_right(
const Packet4i& a) {
1912 return vec_sra(a,
reinterpret_cast<Packet4ui
>(pset1<Packet4i>(N)));
1915EIGEN_STRONG_INLINE Packet4i plogical_shift_right(
const Packet4i& a) {
1916 return vec_sr(a,
reinterpret_cast<Packet4ui
>(pset1<Packet4i>(N)));
1919EIGEN_STRONG_INLINE Packet4i plogical_shift_left(
const Packet4i& a) {
1920 return vec_sl(a,
reinterpret_cast<Packet4ui
>(pset1<Packet4i>(N)));
1923EIGEN_STRONG_INLINE Packet16c parithmetic_shift_right(
const Packet16c& a) {
1924 return vec_sra(a, pset1<Packet16uc>(
static_cast<unsigned char>(N)));
1927EIGEN_STRONG_INLINE Packet16c plogical_shift_right(
const Packet16c& a) {
1928 return vec_sr(a, pset1<Packet16uc>(
static_cast<unsigned char>(N)));
1931EIGEN_STRONG_INLINE Packet16c plogical_shift_left(
const Packet16c& a) {
1932 return vec_sl(a, pset1<Packet16uc>(
static_cast<unsigned char>(N)));
1935EIGEN_STRONG_INLINE Packet16uc parithmetic_shift_right(
const Packet16uc& a) {
1936 return vec_sr(a, pset1<Packet16uc>(
static_cast<unsigned char>(N)));
1939EIGEN_STRONG_INLINE Packet16uc plogical_shift_right(
const Packet16uc& a) {
1940 return vec_sr(a, pset1<Packet16uc>(
static_cast<unsigned char>(N)));
1943EIGEN_STRONG_INLINE Packet16uc plogical_shift_left(
const Packet16uc& a) {
1944 return vec_sl(a, pset1<Packet16uc>(
static_cast<unsigned char>(N)));
1947EIGEN_STRONG_INLINE Packet8s parithmetic_shift_right(
const Packet8s& a) {
1948 const EIGEN_DECLARE_CONST_FAST_Packet8us(mask, N);
1949 return vec_sra(a, p8us_mask);
1952EIGEN_STRONG_INLINE Packet8s plogical_shift_right(
const Packet8s& a) {
1953 const EIGEN_DECLARE_CONST_FAST_Packet8us(mask, N);
1954 return vec_sr(a, p8us_mask);
1957EIGEN_STRONG_INLINE Packet8s plogical_shift_left(
const Packet8s& a) {
1958 const EIGEN_DECLARE_CONST_FAST_Packet8us(mask, N);
1959 return vec_sl(a, p8us_mask);
1962EIGEN_STRONG_INLINE Packet8us parithmetic_shift_right(
const Packet8us& a) {
1963 const EIGEN_DECLARE_CONST_FAST_Packet8us(mask, N);
1964 return vec_sr(a, p8us_mask);
1967EIGEN_STRONG_INLINE Packet8us plogical_shift_right(
const Packet8us& a) {
1968 const EIGEN_DECLARE_CONST_FAST_Packet8us(mask, N);
1969 return vec_sr(a, p8us_mask);
1972EIGEN_STRONG_INLINE Packet8us plogical_shift_left(
const Packet8us& a) {
1973 const EIGEN_DECLARE_CONST_FAST_Packet8us(mask, N);
1974 return vec_sl(a, p8us_mask);
1977EIGEN_STRONG_INLINE Packet4f plogical_shift_left(
const Packet4f& a) {
1978 const EIGEN_DECLARE_CONST_FAST_Packet4ui(mask, N);
1979 Packet4ui r = vec_sl(
reinterpret_cast<Packet4ui
>(a), p4ui_mask);
1980 return reinterpret_cast<Packet4f
>(r);
1984EIGEN_STRONG_INLINE Packet4f plogical_shift_right(
const Packet4f& a) {
1985 const EIGEN_DECLARE_CONST_FAST_Packet4ui(mask, N);
1986 Packet4ui r = vec_sr(
reinterpret_cast<Packet4ui
>(a), p4ui_mask);
1987 return reinterpret_cast<Packet4f
>(r);
1991EIGEN_STRONG_INLINE Packet4ui parithmetic_shift_right(
const Packet4ui& a) {
1992 const EIGEN_DECLARE_CONST_FAST_Packet4ui(mask, N);
1993 return vec_sr(a, p4ui_mask);
1997EIGEN_STRONG_INLINE Packet4ui plogical_shift_right(
const Packet4ui& a) {
1998 const EIGEN_DECLARE_CONST_FAST_Packet4ui(mask, N);
1999 return vec_sr(a, p4ui_mask);
2003EIGEN_STRONG_INLINE Packet4ui plogical_shift_left(
const Packet4ui& a) {
2004 const EIGEN_DECLARE_CONST_FAST_Packet4ui(mask, N);
2005 return vec_sl(a, p4ui_mask);
2008EIGEN_STRONG_INLINE Packet4f Bf16ToF32Even(
const Packet8bf& bf) {
2009 return plogical_shift_left<16>(
reinterpret_cast<Packet4f
>(bf.m_val));
2012EIGEN_STRONG_INLINE Packet4f Bf16ToF32Odd(
const Packet8bf& bf) {
2013 const EIGEN_DECLARE_CONST_FAST_Packet4ui(high_mask, 0xFFFF0000);
2014 return pand<Packet4f>(
reinterpret_cast<Packet4f
>(bf.m_val),
reinterpret_cast<Packet4f
>(p4ui_high_mask));
2017EIGEN_ALWAYS_INLINE Packet8us pmerge(
const Packet4ui& even,
const Packet4ui& odd) {
2019 return vec_perm(
reinterpret_cast<Packet8us
>(odd),
reinterpret_cast<Packet8us
>(even), p16uc_MERGEO16);
2021 return vec_perm(
reinterpret_cast<Packet8us
>(even),
reinterpret_cast<Packet8us
>(odd), p16uc_MERGEE16);
2027EIGEN_STRONG_INLINE Packet8bf F32ToBf16Bool(
const Packet4f& even,
const Packet4f& odd) {
2028 return pmerge(
reinterpret_cast<Packet4ui
>(even),
reinterpret_cast<Packet4ui
>(odd));
2033#ifndef __VEC_CLASS_FP_NAN
2034#define __VEC_CLASS_FP_NAN (1 << 6)
2037#if defined(SUPPORT_BF16_SUBNORMALS) && !defined(__VEC_CLASS_FP_SUBNORMAL)
2038#define __VEC_CLASS_FP_SUBNORMAL_P (1 << 1)
2039#define __VEC_CLASS_FP_SUBNORMAL_N (1 << 0)
2041#define __VEC_CLASS_FP_SUBNORMAL (__VEC_CLASS_FP_SUBNORMAL_P | __VEC_CLASS_FP_SUBNORMAL_N)
2044EIGEN_STRONG_INLINE Packet8bf F32ToBf16(
const Packet4f& p4f) {
2046 return reinterpret_cast<Packet8us
>(__builtin_vsx_xvcvspbf16(
reinterpret_cast<Packet16uc
>(p4f)));
2048 Packet4ui input =
reinterpret_cast<Packet4ui
>(p4f);
2049 Packet4ui lsb = plogical_shift_right<16>(input);
2050 lsb = pand<Packet4ui>(lsb,
reinterpret_cast<Packet4ui
>(p4i_ONE));
2052 EIGEN_DECLARE_CONST_FAST_Packet4ui(BIAS, 0x7FFFu);
2053 Packet4ui rounding_bias = padd<Packet4ui>(lsb, p4ui_BIAS);
2054 input = padd<Packet4ui>(input, rounding_bias);
2056 const EIGEN_DECLARE_CONST_FAST_Packet4ui(nan, 0x7FC00000);
2057#if defined(_ARCH_PWR9) && defined(EIGEN_VECTORIZE_VSX)
2058 Packet4bi nan_selector = vec_test_data_class(p4f, __VEC_CLASS_FP_NAN);
2059 input = vec_sel(input, p4ui_nan, nan_selector);
2061#ifdef SUPPORT_BF16_SUBNORMALS
2062 Packet4bi subnormal_selector = vec_test_data_class(p4f, __VEC_CLASS_FP_SUBNORMAL);
2063 input = vec_sel(input,
reinterpret_cast<Packet4ui
>(p4f), subnormal_selector);
2066#ifdef SUPPORT_BF16_SUBNORMALS
2068 const EIGEN_DECLARE_CONST_FAST_Packet4ui(exp_mask, 0x7F800000);
2069 Packet4ui exp = pand<Packet4ui>(p4ui_exp_mask,
reinterpret_cast<Packet4ui
>(p4f));
2071 const EIGEN_DECLARE_CONST_FAST_Packet4ui(mantissa_mask, 0x7FFFFF);
2072 Packet4ui mantissa = pand<Packet4ui>(p4ui_mantissa_mask,
reinterpret_cast<Packet4ui
>(p4f));
2074 Packet4bi is_max_exp = vec_cmpeq(exp, p4ui_exp_mask);
2075 Packet4bi is_mant_zero = vec_cmpeq(mantissa,
reinterpret_cast<Packet4ui
>(p4i_ZERO));
2077 Packet4ui nan_selector =
2078 pandnot<Packet4ui>(
reinterpret_cast<Packet4ui
>(is_max_exp),
reinterpret_cast<Packet4ui
>(is_mant_zero));
2080 Packet4bi is_zero_exp = vec_cmpeq(exp,
reinterpret_cast<Packet4ui
>(p4i_ZERO));
2082 Packet4ui subnormal_selector =
2083 pandnot<Packet4ui>(
reinterpret_cast<Packet4ui
>(is_zero_exp),
reinterpret_cast<Packet4ui
>(is_mant_zero));
2085 input = vec_sel(input, p4ui_nan, nan_selector);
2086 input = vec_sel(input,
reinterpret_cast<Packet4ui
>(p4f), subnormal_selector);
2089 Packet4bi nan_selector = vec_cmpeq(p4f, p4f);
2091 input = vec_sel(p4ui_nan, input, nan_selector);
2095 input = plogical_shift_right<16>(input);
2096 return reinterpret_cast<Packet8us
>(input);
2107EIGEN_ALWAYS_INLINE Packet8bf Bf16PackHigh(
const Packet4f& lo,
const Packet4f& hi) {
2109 return vec_perm(
reinterpret_cast<Packet8us
>(lo),
reinterpret_cast<Packet8us
>(hi), p16uc_MERGEH16);
2111 return vec_perm(
reinterpret_cast<Packet8us
>(hi),
reinterpret_cast<Packet8us
>(lo), p16uc_MERGEE16);
2121EIGEN_ALWAYS_INLINE Packet8bf Bf16PackLow(
const Packet4f& lo,
const Packet4f& hi) {
2123 return vec_pack(
reinterpret_cast<Packet4ui
>(lo),
reinterpret_cast<Packet4ui
>(hi));
2125 return vec_perm(
reinterpret_cast<Packet8us
>(hi),
reinterpret_cast<Packet8us
>(lo), p16uc_MERGEO16);
2130EIGEN_ALWAYS_INLINE Packet8bf Bf16PackLow(
const Packet4f& hi,
const Packet4f& lo) {
2132 return vec_pack(
reinterpret_cast<Packet4ui
>(hi),
reinterpret_cast<Packet4ui
>(lo));
2134 return vec_perm(
reinterpret_cast<Packet8us
>(hi),
reinterpret_cast<Packet8us
>(lo), p16uc_MERGEE16);
2139EIGEN_ALWAYS_INLINE Packet8bf Bf16PackHigh(
const Packet4f& hi,
const Packet4f& lo) {
2141 return vec_perm(
reinterpret_cast<Packet8us
>(hi),
reinterpret_cast<Packet8us
>(lo), p16uc_MERGEL16);
2143 return vec_perm(
reinterpret_cast<Packet8us
>(hi),
reinterpret_cast<Packet8us
>(lo), p16uc_MERGEO16);
2153template <
bool lohi = true>
2154EIGEN_ALWAYS_INLINE Packet8bf F32ToBf16Two(
const Packet4f& lo,
const Packet4f& hi) {
2155 Packet8us p4f = Bf16PackHigh<lohi>(lo, hi);
2156 Packet8us p4f2 = Bf16PackLow<lohi>(lo, hi);
2158 Packet8us lsb = pand<Packet8us>(p4f, p8us_ONE);
2159 EIGEN_DECLARE_CONST_FAST_Packet8us(BIAS, 0x7FFFu);
2160 lsb = padd<Packet8us>(lsb, p8us_BIAS);
2161 lsb = padd<Packet8us>(lsb, p4f2);
2163 Packet8bi rounding_bias = vec_cmplt(lsb, p4f2);
2164 Packet8us input = psub<Packet8us>(p4f,
reinterpret_cast<Packet8us
>(rounding_bias));
2166#if defined(_ARCH_PWR9) && defined(EIGEN_VECTORIZE_VSX)
2167 Packet4bi nan_selector_lo = vec_test_data_class(lo, __VEC_CLASS_FP_NAN);
2168 Packet4bi nan_selector_hi = vec_test_data_class(hi, __VEC_CLASS_FP_NAN);
2169 Packet8us nan_selector =
2170 Bf16PackLow<lohi>(
reinterpret_cast<Packet4f
>(nan_selector_lo),
reinterpret_cast<Packet4f
>(nan_selector_hi));
2172 input = vec_sel(input, p8us_BIAS, nan_selector);
2174#ifdef SUPPORT_BF16_SUBNORMALS
2175 Packet4bi subnormal_selector_lo = vec_test_data_class(lo, __VEC_CLASS_FP_SUBNORMAL);
2176 Packet4bi subnormal_selector_hi = vec_test_data_class(hi, __VEC_CLASS_FP_SUBNORMAL);
2177 Packet8us subnormal_selector = Bf16PackLow<lohi>(
reinterpret_cast<Packet4f
>(subnormal_selector_lo),
2178 reinterpret_cast<Packet4f
>(subnormal_selector_hi));
2180 input = vec_sel(input,
reinterpret_cast<Packet8us
>(p4f), subnormal_selector);
2183#ifdef SUPPORT_BF16_SUBNORMALS
2185 const EIGEN_DECLARE_CONST_FAST_Packet8us(exp_mask, 0x7F80);
2186 Packet8us exp = pand<Packet8us>(p8us_exp_mask, p4f);
2188 const EIGEN_DECLARE_CONST_FAST_Packet8us(mantissa_mask, 0x7Fu);
2189 Packet8us mantissa = pand<Packet8us>(p8us_mantissa_mask, p4f);
2191 Packet8bi is_max_exp = vec_cmpeq(exp, p8us_exp_mask);
2192 Packet8bi is_mant_zero = vec_cmpeq(mantissa,
reinterpret_cast<Packet8us
>(p4i_ZERO));
2194 Packet8us nan_selector =
2195 pandnot<Packet8us>(
reinterpret_cast<Packet8us
>(is_max_exp),
reinterpret_cast<Packet8us
>(is_mant_zero));
2197 Packet8bi is_zero_exp = vec_cmpeq(exp,
reinterpret_cast<Packet8us
>(p4i_ZERO));
2199 Packet8us subnormal_selector =
2200 pandnot<Packet8us>(
reinterpret_cast<Packet8us
>(is_zero_exp),
reinterpret_cast<Packet8us
>(is_mant_zero));
2203 input = vec_sel(input, p8us_BIAS, nan_selector);
2204 input = vec_sel(input,
reinterpret_cast<Packet8us
>(p4f), subnormal_selector);
2207 Packet4bi nan_selector_lo = vec_cmpeq(lo, lo);
2208 Packet4bi nan_selector_hi = vec_cmpeq(hi, hi);
2209 Packet8us nan_selector =
2210 Bf16PackLow<lohi>(
reinterpret_cast<Packet4f
>(nan_selector_lo),
reinterpret_cast<Packet4f
>(nan_selector_hi));
2212 input = vec_sel(p8us_BIAS, input, nan_selector);
2222EIGEN_STRONG_INLINE Packet8bf F32ToBf16Both(
const Packet4f& lo,
const Packet4f& hi) {
2224 Packet8bf fp16_0 = F32ToBf16(lo);
2225 Packet8bf fp16_1 = F32ToBf16(hi);
2226 return vec_pack(
reinterpret_cast<Packet4ui
>(fp16_0.m_val),
reinterpret_cast<Packet4ui
>(fp16_1.m_val));
2228 return F32ToBf16Two(lo, hi);
2235EIGEN_STRONG_INLINE Packet8bf F32ToBf16(
const Packet4f& even,
const Packet4f& odd) {
2237 return pmerge(
reinterpret_cast<Packet4ui
>(F32ToBf16(even).m_val),
reinterpret_cast<Packet4ui
>(F32ToBf16(odd).m_val));
2239 return F32ToBf16Two<false>(even, odd);
2242#define BF16_TO_F32_UNARY_OP_WRAPPER(OP, A) \
2243 Packet4f a_even = Bf16ToF32Even(A); \
2244 Packet4f a_odd = Bf16ToF32Odd(A); \
2245 Packet4f op_even = OP(a_even); \
2246 Packet4f op_odd = OP(a_odd); \
2247 return F32ToBf16(op_even, op_odd);
2249#define BF16_TO_F32_BINARY_OP_WRAPPER(OP, A, B) \
2250 Packet4f a_even = Bf16ToF32Even(A); \
2251 Packet4f a_odd = Bf16ToF32Odd(A); \
2252 Packet4f b_even = Bf16ToF32Even(B); \
2253 Packet4f b_odd = Bf16ToF32Odd(B); \
2254 Packet4f op_even = OP(a_even, b_even); \
2255 Packet4f op_odd = OP(a_odd, b_odd); \
2256 return F32ToBf16(op_even, op_odd);
2258#define BF16_TO_F32_BINARY_OP_WRAPPER_BOOL(OP, A, B) \
2259 Packet4f a_even = Bf16ToF32Even(A); \
2260 Packet4f a_odd = Bf16ToF32Odd(A); \
2261 Packet4f b_even = Bf16ToF32Even(B); \
2262 Packet4f b_odd = Bf16ToF32Odd(B); \
2263 Packet4f op_even = OP(a_even, b_even); \
2264 Packet4f op_odd = OP(a_odd, b_odd); \
2265 return F32ToBf16Bool(op_even, op_odd);
2268EIGEN_STRONG_INLINE Packet8bf padd<Packet8bf>(
const Packet8bf& a,
const Packet8bf& b) {
2269 BF16_TO_F32_BINARY_OP_WRAPPER(padd<Packet4f>, a, b);
2273EIGEN_STRONG_INLINE Packet8bf pmul<Packet8bf>(
const Packet8bf& a,
const Packet8bf& b) {
2274 BF16_TO_F32_BINARY_OP_WRAPPER(pmul<Packet4f>, a, b);
2278EIGEN_STRONG_INLINE Packet8bf pdiv<Packet8bf>(
const Packet8bf& a,
const Packet8bf& b) {
2279 BF16_TO_F32_BINARY_OP_WRAPPER(pdiv<Packet4f>, a, b);
2283EIGEN_STRONG_INLINE Packet8bf pnegate<Packet8bf>(
const Packet8bf& a) {
2284 EIGEN_DECLARE_CONST_FAST_Packet8us(neg_mask, 0x8000);
2285 return pxor<Packet8us>(p8us_neg_mask, a);
2289EIGEN_STRONG_INLINE Packet8bf psub<Packet8bf>(
const Packet8bf& a,
const Packet8bf& b) {
2290 BF16_TO_F32_BINARY_OP_WRAPPER(psub<Packet4f>, a, b);
2294EIGEN_STRONG_INLINE Packet8bf pexp<Packet8bf>(
const Packet8bf& a) {
2295 BF16_TO_F32_UNARY_OP_WRAPPER(pexp_float, a);
2299EIGEN_STRONG_INLINE Packet8bf pexp2<Packet8bf>(
const Packet8bf& a) {
2300 BF16_TO_F32_UNARY_OP_WRAPPER(generic_exp2, a);
2304EIGEN_STRONG_INLINE Packet4f pldexp<Packet4f>(
const Packet4f& a,
const Packet4f& exponent) {
2305 return pldexp_generic(a, exponent);
2308EIGEN_STRONG_INLINE Packet8bf pldexp<Packet8bf>(
const Packet8bf& a,
const Packet8bf& exponent) {
2309 BF16_TO_F32_BINARY_OP_WRAPPER(pldexp<Packet4f>, a, exponent);
2313EIGEN_STRONG_INLINE Packet4f pfrexp<Packet4f>(
const Packet4f& a, Packet4f& exponent) {
2314 return pfrexp_generic(a, exponent);
2317EIGEN_STRONG_INLINE Packet8bf pfrexp<Packet8bf>(
const Packet8bf& a, Packet8bf& e) {
2318 Packet4f a_even = Bf16ToF32Even(a);
2319 Packet4f a_odd = Bf16ToF32Odd(a);
2322 Packet4f op_even = pfrexp<Packet4f>(a_even, e_even);
2323 Packet4f op_odd = pfrexp<Packet4f>(a_odd, e_odd);
2324 e = F32ToBf16(e_even, e_odd);
2325 return F32ToBf16(op_even, op_odd);
2329EIGEN_STRONG_INLINE Packet8bf psin<Packet8bf>(
const Packet8bf& a) {
2330 BF16_TO_F32_UNARY_OP_WRAPPER(psin_float, a);
2333EIGEN_STRONG_INLINE Packet8bf pcos<Packet8bf>(
const Packet8bf& a) {
2334 BF16_TO_F32_UNARY_OP_WRAPPER(pcos_float, a);
2337EIGEN_STRONG_INLINE Packet8bf plog<Packet8bf>(
const Packet8bf& a) {
2338 BF16_TO_F32_UNARY_OP_WRAPPER(plog_float, a);
2341EIGEN_STRONG_INLINE Packet8bf pfloor<Packet8bf>(
const Packet8bf& a) {
2342 BF16_TO_F32_UNARY_OP_WRAPPER(pfloor<Packet4f>, a);
2345EIGEN_STRONG_INLINE Packet8bf pceil<Packet8bf>(
const Packet8bf& a) {
2346 BF16_TO_F32_UNARY_OP_WRAPPER(pceil<Packet4f>, a);
2349EIGEN_STRONG_INLINE Packet8bf pround<Packet8bf>(
const Packet8bf& a) {
2350 BF16_TO_F32_UNARY_OP_WRAPPER(pround<Packet4f>, a);
2353EIGEN_STRONG_INLINE Packet8bf ptrunc<Packet8bf>(
const Packet8bf& a) {
2354 BF16_TO_F32_UNARY_OP_WRAPPER(ptrunc<Packet4f>, a);
2356#ifdef EIGEN_VECTORIZE_VSX
2358EIGEN_STRONG_INLINE Packet8bf print<Packet8bf>(
const Packet8bf& a) {
2359 BF16_TO_F32_UNARY_OP_WRAPPER(print<Packet4f>, a);
2363EIGEN_STRONG_INLINE Packet8bf pmadd(
const Packet8bf& a,
const Packet8bf& b,
const Packet8bf& c) {
2364 Packet4f a_even = Bf16ToF32Even(a);
2365 Packet4f a_odd = Bf16ToF32Odd(a);
2366 Packet4f b_even = Bf16ToF32Even(b);
2367 Packet4f b_odd = Bf16ToF32Odd(b);
2368 Packet4f c_even = Bf16ToF32Even(c);
2369 Packet4f c_odd = Bf16ToF32Odd(c);
2370 Packet4f pmadd_even = pmadd<Packet4f>(a_even, b_even, c_even);
2371 Packet4f pmadd_odd = pmadd<Packet4f>(a_odd, b_odd, c_odd);
2372 return F32ToBf16(pmadd_even, pmadd_odd);
2376EIGEN_STRONG_INLINE Packet8bf pmsub(
const Packet8bf& a,
const Packet8bf& b,
const Packet8bf& c) {
2377 Packet4f a_even = Bf16ToF32Even(a);
2378 Packet4f a_odd = Bf16ToF32Odd(a);
2379 Packet4f b_even = Bf16ToF32Even(b);
2380 Packet4f b_odd = Bf16ToF32Odd(b);
2381 Packet4f c_even = Bf16ToF32Even(c);
2382 Packet4f c_odd = Bf16ToF32Odd(c);
2383 Packet4f pmadd_even = pmsub<Packet4f>(a_even, b_even, c_even);
2384 Packet4f pmadd_odd = pmsub<Packet4f>(a_odd, b_odd, c_odd);
2385 return F32ToBf16(pmadd_even, pmadd_odd);
2388EIGEN_STRONG_INLINE Packet8bf pnmadd(
const Packet8bf& a,
const Packet8bf& b,
const Packet8bf& c) {
2389 Packet4f a_even = Bf16ToF32Even(a);
2390 Packet4f a_odd = Bf16ToF32Odd(a);
2391 Packet4f b_even = Bf16ToF32Even(b);
2392 Packet4f b_odd = Bf16ToF32Odd(b);
2393 Packet4f c_even = Bf16ToF32Even(c);
2394 Packet4f c_odd = Bf16ToF32Odd(c);
2395 Packet4f pmadd_even = pnmadd<Packet4f>(a_even, b_even, c_even);
2396 Packet4f pmadd_odd = pnmadd<Packet4f>(a_odd, b_odd, c_odd);
2397 return F32ToBf16(pmadd_even, pmadd_odd);
2401EIGEN_STRONG_INLINE Packet8bf pnmsub(
const Packet8bf& a,
const Packet8bf& b,
const Packet8bf& c) {
2402 Packet4f a_even = Bf16ToF32Even(a);
2403 Packet4f a_odd = Bf16ToF32Odd(a);
2404 Packet4f b_even = Bf16ToF32Even(b);
2405 Packet4f b_odd = Bf16ToF32Odd(b);
2406 Packet4f c_even = Bf16ToF32Even(c);
2407 Packet4f c_odd = Bf16ToF32Odd(c);
2408 Packet4f pmadd_even = pnmsub<Packet4f>(a_even, b_even, c_even);
2409 Packet4f pmadd_odd = pnmsub<Packet4f>(a_odd, b_odd, c_odd);
2410 return F32ToBf16(pmadd_even, pmadd_odd);
2414EIGEN_STRONG_INLINE Packet8bf pmin<Packet8bf>(
const Packet8bf& a,
const Packet8bf& b) {
2415 BF16_TO_F32_BINARY_OP_WRAPPER(pmin<Packet4f>, a, b);
2419EIGEN_STRONG_INLINE Packet8bf pmax<Packet8bf>(
const Packet8bf& a,
const Packet8bf& b) {
2420 BF16_TO_F32_BINARY_OP_WRAPPER(pmax<Packet4f>, a, b);
2424EIGEN_STRONG_INLINE Packet8bf pcmp_lt(
const Packet8bf& a,
const Packet8bf& b) {
2425 BF16_TO_F32_BINARY_OP_WRAPPER_BOOL(pcmp_lt<Packet4f>, a, b);
2428EIGEN_STRONG_INLINE Packet8bf pcmp_lt_or_nan(
const Packet8bf& a,
const Packet8bf& b) {
2429 BF16_TO_F32_BINARY_OP_WRAPPER_BOOL(pcmp_lt_or_nan<Packet4f>, a, b);
2432EIGEN_STRONG_INLINE Packet8bf pcmp_le(
const Packet8bf& a,
const Packet8bf& b) {
2433 BF16_TO_F32_BINARY_OP_WRAPPER_BOOL(pcmp_le<Packet4f>, a, b);
2436EIGEN_STRONG_INLINE Packet8bf pcmp_eq(
const Packet8bf& a,
const Packet8bf& b) {
2437 BF16_TO_F32_BINARY_OP_WRAPPER_BOOL(pcmp_eq<Packet4f>, a, b);
2442EIGEN_STRONG_INLINE Packet8bf psign<Packet8bf>(
const Packet8bf& a) {
2443 const Packet8us magnitude = vec_and(a.m_val, vec_splats(
static_cast<unsigned short int>(0x7fff)));
2444 const Packet8bi is_zero = vec_cmpeq(magnitude, vec_splats(
static_cast<unsigned short int>(0)));
2445 const Packet8us keep =
2446 vec_or(
reinterpret_cast<Packet8us
>(vec_cmpgt(magnitude, vec_splats(
static_cast<unsigned short int>(0x7f80)))),
2447 vec_splats(
static_cast<unsigned short int>(0x8000)));
2448 const Packet8us value = vec_sel(vec_splats(
static_cast<unsigned short int>(0x3f80)), a.m_val, keep);
2449 return Packet8bf(vec_andc(value,
reinterpret_cast<Packet8us
>(is_zero)));
2453EIGEN_STRONG_INLINE bfloat16 pfirst(
const Packet8bf& a) {
2454 return Eigen::bfloat16_impl::raw_uint16_to_bfloat16((pfirst<Packet8us>(a)));
2458EIGEN_STRONG_INLINE Packet8bf ploaddup<Packet8bf>(
const bfloat16* from) {
2459 return ploaddup<Packet8us>(
reinterpret_cast<const unsigned short int*
>(from));
2463EIGEN_STRONG_INLINE Packet8bf plset<Packet8bf>(
const bfloat16& a) {
2464 bfloat16 countdown[8] = {bfloat16(0), bfloat16(1), bfloat16(2), bfloat16(3),
2465 bfloat16(4), bfloat16(5), bfloat16(6), bfloat16(7)};
2466 return padd<Packet8bf>(pset1<Packet8bf>(a), pload<Packet8bf>(countdown));
2470EIGEN_STRONG_INLINE
float predux<Packet4f>(
const Packet4f& a) {
2472 b = vec_sld(a, a, 8);
2474 b = vec_sld(sum, sum, 4);
2480EIGEN_STRONG_INLINE
int predux<Packet4i>(
const Packet4i& a) {
2482 b = vec_sld(a, a, 8);
2484 b = vec_sld(sum, sum, 4);
2490EIGEN_STRONG_INLINE bfloat16 predux<Packet8bf>(
const Packet8bf& a) {
2491 float redux_even = predux<Packet4f>(Bf16ToF32Even(a));
2492 float redux_odd = predux<Packet4f>(Bf16ToF32Odd(a));
2493 float f32_result = redux_even + redux_odd;
2494 return bfloat16(f32_result);
2496template <
typename Packet>
2497EIGEN_STRONG_INLINE __UNPACK_TYPE__(Packet) predux_size8(
const Packet& a) {
2500 __UNPACK_TYPE__(Packet) n[8];
2504 EIGEN_ALIGN16
int first_loader[4] = {vt.n[0], vt.n[1], vt.n[2], vt.n[3]};
2505 EIGEN_ALIGN16
int second_loader[4] = {vt.n[4], vt.n[5], vt.n[6], vt.n[7]};
2506 Packet4i first_half = pload<Packet4i>(first_loader);
2507 Packet4i second_half = pload<Packet4i>(second_loader);
2509 return static_cast<__UNPACK_TYPE__(Packet)
>(predux(first_half) + predux(second_half));
2513EIGEN_STRONG_INLINE
short int predux<Packet8s>(
const Packet8s& a) {
2514 return predux_size8<Packet8s>(a);
2518EIGEN_STRONG_INLINE
unsigned short int predux<Packet8us>(
const Packet8us& a) {
2519 return predux_size8<Packet8us>(a);
2522template <
typename Packet>
2523EIGEN_STRONG_INLINE __UNPACK_TYPE__(Packet) predux_size16(
const Packet& a) {
2526 __UNPACK_TYPE__(Packet) n[16];
2530 EIGEN_ALIGN16
int first_loader[4] = {vt.n[0], vt.n[1], vt.n[2], vt.n[3]};
2531 EIGEN_ALIGN16
int second_loader[4] = {vt.n[4], vt.n[5], vt.n[6], vt.n[7]};
2532 EIGEN_ALIGN16
int third_loader[4] = {vt.n[8], vt.n[9], vt.n[10], vt.n[11]};
2533 EIGEN_ALIGN16
int fourth_loader[4] = {vt.n[12], vt.n[13], vt.n[14], vt.n[15]};
2535 Packet4i first_quarter = pload<Packet4i>(first_loader);
2536 Packet4i second_quarter = pload<Packet4i>(second_loader);
2537 Packet4i third_quarter = pload<Packet4i>(third_loader);
2538 Packet4i fourth_quarter = pload<Packet4i>(fourth_loader);
2540 return static_cast<__UNPACK_TYPE__(Packet)
>(predux(first_quarter) + predux(second_quarter) + predux(third_quarter) +
2541 predux(fourth_quarter));
2545EIGEN_STRONG_INLINE
signed char predux<Packet16c>(
const Packet16c& a) {
2546 return predux_size16<Packet16c>(a);
2550EIGEN_STRONG_INLINE
unsigned char predux<Packet16uc>(
const Packet16uc& a) {
2551 return predux_size16<Packet16uc>(a);
2557EIGEN_STRONG_INLINE
float predux_mul<Packet4f>(
const Packet4f& a) {
2559 prod = pmul(a, vec_sld(a, a, 8));
2560 return pfirst(pmul(prod, vec_sld(prod, prod, 4)));
2564EIGEN_STRONG_INLINE
int predux_mul<Packet4i>(
const Packet4i& a) {
2565 EIGEN_ALIGN16
int aux[4];
2567 return aux[0] * aux[1] * aux[2] * aux[3];
2571EIGEN_STRONG_INLINE
short int predux_mul<Packet8s>(
const Packet8s& a) {
2572 Packet8s pair, quad, octo;
2574 pair = vec_mul(a, vec_sld(a, a, 8));
2575 quad = vec_mul(pair, vec_sld(pair, pair, 4));
2576 octo = vec_mul(quad, vec_sld(quad, quad, 2));
2578 return pfirst(octo);
2582EIGEN_STRONG_INLINE
unsigned short int predux_mul<Packet8us>(
const Packet8us& a) {
2583 Packet8us pair, quad, octo;
2585 pair = vec_mul(a, vec_sld(a, a, 8));
2586 quad = vec_mul(pair, vec_sld(pair, pair, 4));
2587 octo = vec_mul(quad, vec_sld(quad, quad, 2));
2589 return pfirst(octo);
2593EIGEN_STRONG_INLINE bfloat16 predux_mul<Packet8bf>(
const Packet8bf& a) {
2594 float redux_even = predux_mul<Packet4f>(Bf16ToF32Even(a));
2595 float redux_odd = predux_mul<Packet4f>(Bf16ToF32Odd(a));
2596 float f32_result = redux_even * redux_odd;
2597 return bfloat16(f32_result);
2601EIGEN_STRONG_INLINE
signed char predux_mul<Packet16c>(
const Packet16c& a) {
2602 Packet16c pair, quad, octo, result;
2604 pair = vec_mul(a, vec_sld(a, a, 8));
2605 quad = vec_mul(pair, vec_sld(pair, pair, 4));
2606 octo = vec_mul(quad, vec_sld(quad, quad, 2));
2607 result = vec_mul(octo, vec_sld(octo, octo, 1));
2609 return pfirst(result);
2613EIGEN_STRONG_INLINE
unsigned char predux_mul<Packet16uc>(
const Packet16uc& a) {
2614 Packet16uc pair, quad, octo, result;
2616 pair = vec_mul(a, vec_sld(a, a, 8));
2617 quad = vec_mul(pair, vec_sld(pair, pair, 4));
2618 octo = vec_mul(quad, vec_sld(quad, quad, 2));
2619 result = vec_mul(octo, vec_sld(octo, octo, 1));
2621 return pfirst(result);
2625template <
typename Packet>
2626EIGEN_STRONG_INLINE __UNPACK_TYPE__(Packet) predux_min4(
const Packet& a) {
2628 b = vec_min(a, vec_sld(a, a, 8));
2629 res = vec_min(b, vec_sld(b, b, 4));
2634EIGEN_STRONG_INLINE
float predux_min<Packet4f>(
const Packet4f& a) {
2635 return predux_min4<Packet4f>(a);
2639EIGEN_STRONG_INLINE
int predux_min<Packet4i>(
const Packet4i& a) {
2640 return predux_min4<Packet4i>(a);
2644EIGEN_STRONG_INLINE bfloat16 predux_min<Packet8bf>(
const Packet8bf& a) {
2645 float redux_even = predux_min<Packet4f>(Bf16ToF32Even(a));
2646 float redux_odd = predux_min<Packet4f>(Bf16ToF32Odd(a));
2647 float f32_result = (std::min)(redux_even, redux_odd);
2648 return bfloat16(f32_result);
2652EIGEN_STRONG_INLINE
short int predux_min<Packet8s>(
const Packet8s& a) {
2653 Packet8s pair, quad, octo;
2656 pair = vec_min(a, vec_sld(a, a, 8));
2659 quad = vec_min(pair, vec_sld(pair, pair, 4));
2662 octo = vec_min(quad, vec_sld(quad, quad, 2));
2663 return pfirst(octo);
2667EIGEN_STRONG_INLINE
unsigned short int predux_min<Packet8us>(
const Packet8us& a) {
2668 Packet8us pair, quad, octo;
2671 pair = vec_min(a, vec_sld(a, a, 8));
2674 quad = vec_min(pair, vec_sld(pair, pair, 4));
2677 octo = vec_min(quad, vec_sld(quad, quad, 2));
2678 return pfirst(octo);
2682EIGEN_STRONG_INLINE
signed char predux_min<Packet16c>(
const Packet16c& a) {
2683 Packet16c pair, quad, octo, result;
2685 pair = vec_min(a, vec_sld(a, a, 8));
2686 quad = vec_min(pair, vec_sld(pair, pair, 4));
2687 octo = vec_min(quad, vec_sld(quad, quad, 2));
2688 result = vec_min(octo, vec_sld(octo, octo, 1));
2690 return pfirst(result);
2694EIGEN_STRONG_INLINE
unsigned char predux_min<Packet16uc>(
const Packet16uc& a) {
2695 Packet16uc pair, quad, octo, result;
2697 pair = vec_min(a, vec_sld(a, a, 8));
2698 quad = vec_min(pair, vec_sld(pair, pair, 4));
2699 octo = vec_min(quad, vec_sld(quad, quad, 2));
2700 result = vec_min(octo, vec_sld(octo, octo, 1));
2702 return pfirst(result);
2705template <
typename Packet>
2706EIGEN_STRONG_INLINE __UNPACK_TYPE__(Packet) predux_max4(
const Packet& a) {
2708 b = vec_max(a, vec_sld(a, a, 8));
2709 res = vec_max(b, vec_sld(b, b, 4));
2714EIGEN_STRONG_INLINE
float predux_max<Packet4f>(
const Packet4f& a) {
2715 return predux_max4<Packet4f>(a);
2719EIGEN_STRONG_INLINE
int predux_max<Packet4i>(
const Packet4i& a) {
2720 return predux_max4<Packet4i>(a);
2724EIGEN_STRONG_INLINE bfloat16 predux_max<Packet8bf>(
const Packet8bf& a) {
2725 float redux_even = predux_max<Packet4f>(Bf16ToF32Even(a));
2726 float redux_odd = predux_max<Packet4f>(Bf16ToF32Odd(a));
2727 float f32_result = (std::max)(redux_even, redux_odd);
2728 return bfloat16(f32_result);
2732EIGEN_STRONG_INLINE
short int predux_max<Packet8s>(
const Packet8s& a) {
2733 Packet8s pair, quad, octo;
2736 pair = vec_max(a, vec_sld(a, a, 8));
2739 quad = vec_max(pair, vec_sld(pair, pair, 4));
2742 octo = vec_max(quad, vec_sld(quad, quad, 2));
2743 return pfirst(octo);
2747EIGEN_STRONG_INLINE
unsigned short int predux_max<Packet8us>(
const Packet8us& a) {
2748 Packet8us pair, quad, octo;
2751 pair = vec_max(a, vec_sld(a, a, 8));
2754 quad = vec_max(pair, vec_sld(pair, pair, 4));
2757 octo = vec_max(quad, vec_sld(quad, quad, 2));
2758 return pfirst(octo);
2762EIGEN_STRONG_INLINE
signed char predux_max<Packet16c>(
const Packet16c& a) {
2763 Packet16c pair, quad, octo, result;
2765 pair = vec_max(a, vec_sld(a, a, 8));
2766 quad = vec_max(pair, vec_sld(pair, pair, 4));
2767 octo = vec_max(quad, vec_sld(quad, quad, 2));
2768 result = vec_max(octo, vec_sld(octo, octo, 1));
2770 return pfirst(result);
2774EIGEN_STRONG_INLINE
unsigned char predux_max<Packet16uc>(
const Packet16uc& a) {
2775 Packet16uc pair, quad, octo, result;
2777 pair = vec_max(a, vec_sld(a, a, 8));
2778 quad = vec_max(pair, vec_sld(pair, pair, 4));
2779 octo = vec_max(quad, vec_sld(quad, quad, 2));
2780 result = vec_max(octo, vec_sld(octo, octo, 1));
2782 return pfirst(result);
2786EIGEN_STRONG_INLINE
bool predux_any(
const Packet4f& x) {
2787 const Packet4ui zero = {0, 0, 0, 0};
2788 return vec_any_ne(
reinterpret_cast<Packet4ui
>(x), zero);
2792EIGEN_STRONG_INLINE
bool predux_any(
const Packet8bf& x) {
2793 const Packet8us zero = {0, 0, 0, 0, 0, 0, 0, 0};
2794 return vec_any_ne(x.m_val, zero);
2797template <
typename T>
2798EIGEN_DEVICE_FUNC
inline void ptranpose_common(PacketBlock<T, 4>& kernel) {
2800 t0 = vec_mergeh(kernel.packet[0], kernel.packet[2]);
2801 t1 = vec_mergel(kernel.packet[0], kernel.packet[2]);
2802 t2 = vec_mergeh(kernel.packet[1], kernel.packet[3]);
2803 t3 = vec_mergel(kernel.packet[1], kernel.packet[3]);
2804 kernel.packet[0] = vec_mergeh(t0, t2);
2805 kernel.packet[1] = vec_mergel(t0, t2);
2806 kernel.packet[2] = vec_mergeh(t1, t3);
2807 kernel.packet[3] = vec_mergel(t1, t3);
2810EIGEN_DEVICE_FUNC
inline void ptranspose(PacketBlock<Packet4f, 4>& kernel) { ptranpose_common<Packet4f>(kernel); }
2812EIGEN_DEVICE_FUNC
inline void ptranspose(PacketBlock<Packet4i, 4>& kernel) { ptranpose_common<Packet4i>(kernel); }
2814EIGEN_DEVICE_FUNC
inline void ptranspose(PacketBlock<Packet8s, 4>& kernel) {
2815 Packet8s t0, t1, t2, t3;
2816 t0 = vec_mergeh(kernel.packet[0], kernel.packet[2]);
2817 t1 = vec_mergel(kernel.packet[0], kernel.packet[2]);
2818 t2 = vec_mergeh(kernel.packet[1], kernel.packet[3]);
2819 t3 = vec_mergel(kernel.packet[1], kernel.packet[3]);
2820 kernel.packet[0] = vec_mergeh(t0, t2);
2821 kernel.packet[1] = vec_mergel(t0, t2);
2822 kernel.packet[2] = vec_mergeh(t1, t3);
2823 kernel.packet[3] = vec_mergel(t1, t3);
2826EIGEN_DEVICE_FUNC
inline void ptranspose(PacketBlock<Packet8us, 4>& kernel) {
2827 Packet8us t0, t1, t2, t3;
2828 t0 = vec_mergeh(kernel.packet[0], kernel.packet[2]);
2829 t1 = vec_mergel(kernel.packet[0], kernel.packet[2]);
2830 t2 = vec_mergeh(kernel.packet[1], kernel.packet[3]);
2831 t3 = vec_mergel(kernel.packet[1], kernel.packet[3]);
2832 kernel.packet[0] = vec_mergeh(t0, t2);
2833 kernel.packet[1] = vec_mergel(t0, t2);
2834 kernel.packet[2] = vec_mergeh(t1, t3);
2835 kernel.packet[3] = vec_mergel(t1, t3);
2838EIGEN_DEVICE_FUNC
inline void ptranspose(PacketBlock<Packet8bf, 4>& kernel) {
2839 Packet8us t0, t1, t2, t3;
2841 t0 = vec_mergeh(kernel.packet[0].m_val, kernel.packet[2].m_val);
2842 t1 = vec_mergel(kernel.packet[0].m_val, kernel.packet[2].m_val);
2843 t2 = vec_mergeh(kernel.packet[1].m_val, kernel.packet[3].m_val);
2844 t3 = vec_mergel(kernel.packet[1].m_val, kernel.packet[3].m_val);
2845 kernel.packet[0] = vec_mergeh(t0, t2);
2846 kernel.packet[1] = vec_mergel(t0, t2);
2847 kernel.packet[2] = vec_mergeh(t1, t3);
2848 kernel.packet[3] = vec_mergel(t1, t3);
2851EIGEN_DEVICE_FUNC
inline void ptranspose(PacketBlock<Packet16c, 4>& kernel) {
2852 Packet16c t0, t1, t2, t3;
2853 t0 = vec_mergeh(kernel.packet[0], kernel.packet[2]);
2854 t1 = vec_mergel(kernel.packet[0], kernel.packet[2]);
2855 t2 = vec_mergeh(kernel.packet[1], kernel.packet[3]);
2856 t3 = vec_mergel(kernel.packet[1], kernel.packet[3]);
2857 kernel.packet[0] = vec_mergeh(t0, t2);
2858 kernel.packet[1] = vec_mergel(t0, t2);
2859 kernel.packet[2] = vec_mergeh(t1, t3);
2860 kernel.packet[3] = vec_mergel(t1, t3);
2863EIGEN_DEVICE_FUNC
inline void ptranspose(PacketBlock<Packet16uc, 4>& kernel) {
2864 Packet16uc t0, t1, t2, t3;
2865 t0 = vec_mergeh(kernel.packet[0], kernel.packet[2]);
2866 t1 = vec_mergel(kernel.packet[0], kernel.packet[2]);
2867 t2 = vec_mergeh(kernel.packet[1], kernel.packet[3]);
2868 t3 = vec_mergel(kernel.packet[1], kernel.packet[3]);
2869 kernel.packet[0] = vec_mergeh(t0, t2);
2870 kernel.packet[1] = vec_mergel(t0, t2);
2871 kernel.packet[2] = vec_mergeh(t1, t3);
2872 kernel.packet[3] = vec_mergel(t1, t3);
2875EIGEN_DEVICE_FUNC
inline void ptranspose(PacketBlock<Packet8s, 8>& kernel) {
2876 Packet8s v[8], sum[8];
2878 v[0] = vec_mergeh(kernel.packet[0], kernel.packet[4]);
2879 v[1] = vec_mergel(kernel.packet[0], kernel.packet[4]);
2880 v[2] = vec_mergeh(kernel.packet[1], kernel.packet[5]);
2881 v[3] = vec_mergel(kernel.packet[1], kernel.packet[5]);
2882 v[4] = vec_mergeh(kernel.packet[2], kernel.packet[6]);
2883 v[5] = vec_mergel(kernel.packet[2], kernel.packet[6]);
2884 v[6] = vec_mergeh(kernel.packet[3], kernel.packet[7]);
2885 v[7] = vec_mergel(kernel.packet[3], kernel.packet[7]);
2886 sum[0] = vec_mergeh(v[0], v[4]);
2887 sum[1] = vec_mergel(v[0], v[4]);
2888 sum[2] = vec_mergeh(v[1], v[5]);
2889 sum[3] = vec_mergel(v[1], v[5]);
2890 sum[4] = vec_mergeh(v[2], v[6]);
2891 sum[5] = vec_mergel(v[2], v[6]);
2892 sum[6] = vec_mergeh(v[3], v[7]);
2893 sum[7] = vec_mergel(v[3], v[7]);
2895 kernel.packet[0] = vec_mergeh(sum[0], sum[4]);
2896 kernel.packet[1] = vec_mergel(sum[0], sum[4]);
2897 kernel.packet[2] = vec_mergeh(sum[1], sum[5]);
2898 kernel.packet[3] = vec_mergel(sum[1], sum[5]);
2899 kernel.packet[4] = vec_mergeh(sum[2], sum[6]);
2900 kernel.packet[5] = vec_mergel(sum[2], sum[6]);
2901 kernel.packet[6] = vec_mergeh(sum[3], sum[7]);
2902 kernel.packet[7] = vec_mergel(sum[3], sum[7]);
2905EIGEN_DEVICE_FUNC
inline void ptranspose(PacketBlock<Packet8us, 8>& kernel) {
2906 Packet8us v[8], sum[8];
2908 v[0] = vec_mergeh(kernel.packet[0], kernel.packet[4]);
2909 v[1] = vec_mergel(kernel.packet[0], kernel.packet[4]);
2910 v[2] = vec_mergeh(kernel.packet[1], kernel.packet[5]);
2911 v[3] = vec_mergel(kernel.packet[1], kernel.packet[5]);
2912 v[4] = vec_mergeh(kernel.packet[2], kernel.packet[6]);
2913 v[5] = vec_mergel(kernel.packet[2], kernel.packet[6]);
2914 v[6] = vec_mergeh(kernel.packet[3], kernel.packet[7]);
2915 v[7] = vec_mergel(kernel.packet[3], kernel.packet[7]);
2916 sum[0] = vec_mergeh(v[0], v[4]);
2917 sum[1] = vec_mergel(v[0], v[4]);
2918 sum[2] = vec_mergeh(v[1], v[5]);
2919 sum[3] = vec_mergel(v[1], v[5]);
2920 sum[4] = vec_mergeh(v[2], v[6]);
2921 sum[5] = vec_mergel(v[2], v[6]);
2922 sum[6] = vec_mergeh(v[3], v[7]);
2923 sum[7] = vec_mergel(v[3], v[7]);
2925 kernel.packet[0] = vec_mergeh(sum[0], sum[4]);
2926 kernel.packet[1] = vec_mergel(sum[0], sum[4]);
2927 kernel.packet[2] = vec_mergeh(sum[1], sum[5]);
2928 kernel.packet[3] = vec_mergel(sum[1], sum[5]);
2929 kernel.packet[4] = vec_mergeh(sum[2], sum[6]);
2930 kernel.packet[5] = vec_mergel(sum[2], sum[6]);
2931 kernel.packet[6] = vec_mergeh(sum[3], sum[7]);
2932 kernel.packet[7] = vec_mergel(sum[3], sum[7]);
2935EIGEN_DEVICE_FUNC
inline void ptranspose(PacketBlock<Packet8bf, 8>& kernel) {
2936 Packet8bf v[8], sum[8];
2938 v[0] = vec_mergeh(kernel.packet[0].m_val, kernel.packet[4].m_val);
2939 v[1] = vec_mergel(kernel.packet[0].m_val, kernel.packet[4].m_val);
2940 v[2] = vec_mergeh(kernel.packet[1].m_val, kernel.packet[5].m_val);
2941 v[3] = vec_mergel(kernel.packet[1].m_val, kernel.packet[5].m_val);
2942 v[4] = vec_mergeh(kernel.packet[2].m_val, kernel.packet[6].m_val);
2943 v[5] = vec_mergel(kernel.packet[2].m_val, kernel.packet[6].m_val);
2944 v[6] = vec_mergeh(kernel.packet[3].m_val, kernel.packet[7].m_val);
2945 v[7] = vec_mergel(kernel.packet[3].m_val, kernel.packet[7].m_val);
2946 sum[0] = vec_mergeh(v[0].m_val, v[4].m_val);
2947 sum[1] = vec_mergel(v[0].m_val, v[4].m_val);
2948 sum[2] = vec_mergeh(v[1].m_val, v[5].m_val);
2949 sum[3] = vec_mergel(v[1].m_val, v[5].m_val);
2950 sum[4] = vec_mergeh(v[2].m_val, v[6].m_val);
2951 sum[5] = vec_mergel(v[2].m_val, v[6].m_val);
2952 sum[6] = vec_mergeh(v[3].m_val, v[7].m_val);
2953 sum[7] = vec_mergel(v[3].m_val, v[7].m_val);
2955 kernel.packet[0] = vec_mergeh(sum[0].m_val, sum[4].m_val);
2956 kernel.packet[1] = vec_mergel(sum[0].m_val, sum[4].m_val);
2957 kernel.packet[2] = vec_mergeh(sum[1].m_val, sum[5].m_val);
2958 kernel.packet[3] = vec_mergel(sum[1].m_val, sum[5].m_val);
2959 kernel.packet[4] = vec_mergeh(sum[2].m_val, sum[6].m_val);
2960 kernel.packet[5] = vec_mergel(sum[2].m_val, sum[6].m_val);
2961 kernel.packet[6] = vec_mergeh(sum[3].m_val, sum[7].m_val);
2962 kernel.packet[7] = vec_mergel(sum[3].m_val, sum[7].m_val);
2965EIGEN_DEVICE_FUNC
inline void ptranspose(PacketBlock<Packet16c, 16>& kernel) {
2966 Packet16c step1[16], step2[16], step3[16];
2968 step1[0] = vec_mergeh(kernel.packet[0], kernel.packet[8]);
2969 step1[1] = vec_mergel(kernel.packet[0], kernel.packet[8]);
2970 step1[2] = vec_mergeh(kernel.packet[1], kernel.packet[9]);
2971 step1[3] = vec_mergel(kernel.packet[1], kernel.packet[9]);
2972 step1[4] = vec_mergeh(kernel.packet[2], kernel.packet[10]);
2973 step1[5] = vec_mergel(kernel.packet[2], kernel.packet[10]);
2974 step1[6] = vec_mergeh(kernel.packet[3], kernel.packet[11]);
2975 step1[7] = vec_mergel(kernel.packet[3], kernel.packet[11]);
2976 step1[8] = vec_mergeh(kernel.packet[4], kernel.packet[12]);
2977 step1[9] = vec_mergel(kernel.packet[4], kernel.packet[12]);
2978 step1[10] = vec_mergeh(kernel.packet[5], kernel.packet[13]);
2979 step1[11] = vec_mergel(kernel.packet[5], kernel.packet[13]);
2980 step1[12] = vec_mergeh(kernel.packet[6], kernel.packet[14]);
2981 step1[13] = vec_mergel(kernel.packet[6], kernel.packet[14]);
2982 step1[14] = vec_mergeh(kernel.packet[7], kernel.packet[15]);
2983 step1[15] = vec_mergel(kernel.packet[7], kernel.packet[15]);
2985 step2[0] = vec_mergeh(step1[0], step1[8]);
2986 step2[1] = vec_mergel(step1[0], step1[8]);
2987 step2[2] = vec_mergeh(step1[1], step1[9]);
2988 step2[3] = vec_mergel(step1[1], step1[9]);
2989 step2[4] = vec_mergeh(step1[2], step1[10]);
2990 step2[5] = vec_mergel(step1[2], step1[10]);
2991 step2[6] = vec_mergeh(step1[3], step1[11]);
2992 step2[7] = vec_mergel(step1[3], step1[11]);
2993 step2[8] = vec_mergeh(step1[4], step1[12]);
2994 step2[9] = vec_mergel(step1[4], step1[12]);
2995 step2[10] = vec_mergeh(step1[5], step1[13]);
2996 step2[11] = vec_mergel(step1[5], step1[13]);
2997 step2[12] = vec_mergeh(step1[6], step1[14]);
2998 step2[13] = vec_mergel(step1[6], step1[14]);
2999 step2[14] = vec_mergeh(step1[7], step1[15]);
3000 step2[15] = vec_mergel(step1[7], step1[15]);
3002 step3[0] = vec_mergeh(step2[0], step2[8]);
3003 step3[1] = vec_mergel(step2[0], step2[8]);
3004 step3[2] = vec_mergeh(step2[1], step2[9]);
3005 step3[3] = vec_mergel(step2[1], step2[9]);
3006 step3[4] = vec_mergeh(step2[2], step2[10]);
3007 step3[5] = vec_mergel(step2[2], step2[10]);
3008 step3[6] = vec_mergeh(step2[3], step2[11]);
3009 step3[7] = vec_mergel(step2[3], step2[11]);
3010 step3[8] = vec_mergeh(step2[4], step2[12]);
3011 step3[9] = vec_mergel(step2[4], step2[12]);
3012 step3[10] = vec_mergeh(step2[5], step2[13]);
3013 step3[11] = vec_mergel(step2[5], step2[13]);
3014 step3[12] = vec_mergeh(step2[6], step2[14]);
3015 step3[13] = vec_mergel(step2[6], step2[14]);
3016 step3[14] = vec_mergeh(step2[7], step2[15]);
3017 step3[15] = vec_mergel(step2[7], step2[15]);
3019 kernel.packet[0] = vec_mergeh(step3[0], step3[8]);
3020 kernel.packet[1] = vec_mergel(step3[0], step3[8]);
3021 kernel.packet[2] = vec_mergeh(step3[1], step3[9]);
3022 kernel.packet[3] = vec_mergel(step3[1], step3[9]);
3023 kernel.packet[4] = vec_mergeh(step3[2], step3[10]);
3024 kernel.packet[5] = vec_mergel(step3[2], step3[10]);
3025 kernel.packet[6] = vec_mergeh(step3[3], step3[11]);
3026 kernel.packet[7] = vec_mergel(step3[3], step3[11]);
3027 kernel.packet[8] = vec_mergeh(step3[4], step3[12]);
3028 kernel.packet[9] = vec_mergel(step3[4], step3[12]);
3029 kernel.packet[10] = vec_mergeh(step3[5], step3[13]);
3030 kernel.packet[11] = vec_mergel(step3[5], step3[13]);
3031 kernel.packet[12] = vec_mergeh(step3[6], step3[14]);
3032 kernel.packet[13] = vec_mergel(step3[6], step3[14]);
3033 kernel.packet[14] = vec_mergeh(step3[7], step3[15]);
3034 kernel.packet[15] = vec_mergel(step3[7], step3[15]);
3037EIGEN_DEVICE_FUNC
inline void ptranspose(PacketBlock<Packet16uc, 16>& kernel) {
3038 Packet16uc step1[16], step2[16], step3[16];
3040 step1[0] = vec_mergeh(kernel.packet[0], kernel.packet[8]);
3041 step1[1] = vec_mergel(kernel.packet[0], kernel.packet[8]);
3042 step1[2] = vec_mergeh(kernel.packet[1], kernel.packet[9]);
3043 step1[3] = vec_mergel(kernel.packet[1], kernel.packet[9]);
3044 step1[4] = vec_mergeh(kernel.packet[2], kernel.packet[10]);
3045 step1[5] = vec_mergel(kernel.packet[2], kernel.packet[10]);
3046 step1[6] = vec_mergeh(kernel.packet[3], kernel.packet[11]);
3047 step1[7] = vec_mergel(kernel.packet[3], kernel.packet[11]);
3048 step1[8] = vec_mergeh(kernel.packet[4], kernel.packet[12]);
3049 step1[9] = vec_mergel(kernel.packet[4], kernel.packet[12]);
3050 step1[10] = vec_mergeh(kernel.packet[5], kernel.packet[13]);
3051 step1[11] = vec_mergel(kernel.packet[5], kernel.packet[13]);
3052 step1[12] = vec_mergeh(kernel.packet[6], kernel.packet[14]);
3053 step1[13] = vec_mergel(kernel.packet[6], kernel.packet[14]);
3054 step1[14] = vec_mergeh(kernel.packet[7], kernel.packet[15]);
3055 step1[15] = vec_mergel(kernel.packet[7], kernel.packet[15]);
3057 step2[0] = vec_mergeh(step1[0], step1[8]);
3058 step2[1] = vec_mergel(step1[0], step1[8]);
3059 step2[2] = vec_mergeh(step1[1], step1[9]);
3060 step2[3] = vec_mergel(step1[1], step1[9]);
3061 step2[4] = vec_mergeh(step1[2], step1[10]);
3062 step2[5] = vec_mergel(step1[2], step1[10]);
3063 step2[6] = vec_mergeh(step1[3], step1[11]);
3064 step2[7] = vec_mergel(step1[3], step1[11]);
3065 step2[8] = vec_mergeh(step1[4], step1[12]);
3066 step2[9] = vec_mergel(step1[4], step1[12]);
3067 step2[10] = vec_mergeh(step1[5], step1[13]);
3068 step2[11] = vec_mergel(step1[5], step1[13]);
3069 step2[12] = vec_mergeh(step1[6], step1[14]);
3070 step2[13] = vec_mergel(step1[6], step1[14]);
3071 step2[14] = vec_mergeh(step1[7], step1[15]);
3072 step2[15] = vec_mergel(step1[7], step1[15]);
3074 step3[0] = vec_mergeh(step2[0], step2[8]);
3075 step3[1] = vec_mergel(step2[0], step2[8]);
3076 step3[2] = vec_mergeh(step2[1], step2[9]);
3077 step3[3] = vec_mergel(step2[1], step2[9]);
3078 step3[4] = vec_mergeh(step2[2], step2[10]);
3079 step3[5] = vec_mergel(step2[2], step2[10]);
3080 step3[6] = vec_mergeh(step2[3], step2[11]);
3081 step3[7] = vec_mergel(step2[3], step2[11]);
3082 step3[8] = vec_mergeh(step2[4], step2[12]);
3083 step3[9] = vec_mergel(step2[4], step2[12]);
3084 step3[10] = vec_mergeh(step2[5], step2[13]);
3085 step3[11] = vec_mergel(step2[5], step2[13]);
3086 step3[12] = vec_mergeh(step2[6], step2[14]);
3087 step3[13] = vec_mergel(step2[6], step2[14]);
3088 step3[14] = vec_mergeh(step2[7], step2[15]);
3089 step3[15] = vec_mergel(step2[7], step2[15]);
3091 kernel.packet[0] = vec_mergeh(step3[0], step3[8]);
3092 kernel.packet[1] = vec_mergel(step3[0], step3[8]);
3093 kernel.packet[2] = vec_mergeh(step3[1], step3[9]);
3094 kernel.packet[3] = vec_mergel(step3[1], step3[9]);
3095 kernel.packet[4] = vec_mergeh(step3[2], step3[10]);
3096 kernel.packet[5] = vec_mergel(step3[2], step3[10]);
3097 kernel.packet[6] = vec_mergeh(step3[3], step3[11]);
3098 kernel.packet[7] = vec_mergel(step3[3], step3[11]);
3099 kernel.packet[8] = vec_mergeh(step3[4], step3[12]);
3100 kernel.packet[9] = vec_mergel(step3[4], step3[12]);
3101 kernel.packet[10] = vec_mergeh(step3[5], step3[13]);
3102 kernel.packet[11] = vec_mergel(step3[5], step3[13]);
3103 kernel.packet[12] = vec_mergeh(step3[6], step3[14]);
3104 kernel.packet[13] = vec_mergel(step3[6], step3[14]);
3105 kernel.packet[14] = vec_mergeh(step3[7], step3[15]);
3106 kernel.packet[15] = vec_mergel(step3[7], step3[15]);
3110#ifdef EIGEN_VECTORIZE_VSX
3111typedef __vector
double Packet2d;
3112typedef __vector
unsigned long long Packet2ul;
3113typedef __vector
long long Packet2l;
3115typedef Packet2ul Packet2bl;
3117typedef __vector __bool
long Packet2bl;
3120static Packet2l p2l_ZERO =
reinterpret_cast<Packet2l
>(p4i_ZERO);
3121static Packet2ul p2ul_SIGN = {0x8000000000000000ull, 0x8000000000000000ull};
3122static Packet2ul p2ul_PREV0DOT5 = {0x3FDFFFFFFFFFFFFFull, 0x3FDFFFFFFFFFFFFFull};
3123static Packet2d p2d_ONE = {1.0, 1.0};
3124static Packet2d p2d_ZERO =
reinterpret_cast<Packet2d
>(p4f_ZERO);
3125static Packet2d p2d_MZERO = {numext::bit_cast<double>(0x8000000000000000ull),
3126 numext::bit_cast<double>(0x8000000000000000ull)};
3129static Packet2d p2d_COUNTDOWN =
3130 reinterpret_cast<Packet2d
>(vec_sld(
reinterpret_cast<Packet4f
>(p2d_ZERO),
reinterpret_cast<Packet4f
>(p2d_ONE), 8));
3132static Packet2d p2d_COUNTDOWN =
3133 reinterpret_cast<Packet2d
>(vec_sld(
reinterpret_cast<Packet4f
>(p2d_ONE),
reinterpret_cast<Packet4f
>(p2d_ZERO), 8));
3137Packet2d vec_splat_dbl(
const Packet2d& a) {
3138 return vec_splat(a, index);
3142struct packet_traits<double> : default_packet_traits {
3143 typedef Packet2d type;
3144 typedef Packet2d half;
3147 AlignedOnScalar = 1,
3157 HasSin = EIGEN_FAST_MATH,
3158 HasCos = EIGEN_FAST_MATH,
3159 HasTan = EIGEN_FAST_MATH,
3160 HasTanh = EIGEN_FAST_MATH,
3161 HasErf = EIGEN_FAST_MATH,
3162 HasErfc = EIGEN_FAST_MATH,
3172#if !EIGEN_COMP_CLANG
3182struct unpacket_traits<Packet2d> {
3183 typedef double type;
3184 typedef Packet2l integer_packet;
3188 vectorizable =
true,
3189 masked_load_available =
false,
3190 masked_store_available =
false
3192 typedef Packet2d half;
3195struct unpacket_traits<Packet2l> {
3196 typedef int64_t type;
3197 typedef Packet2l half;
3201 vectorizable =
false,
3202 masked_load_available =
false,
3203 masked_store_available =
false
3207inline std::ostream& operator<<(std::ostream& s,
const Packet2l& v) {
3213 s << vt.n[0] <<
", " << vt.n[1];
3217inline std::ostream& operator<<(std::ostream& s,
const Packet2d& v) {
3223 s << vt.n[0] <<
", " << vt.n[1];
3229EIGEN_STRONG_INLINE Packet2d pload<Packet2d>(
const double* from) {
3230 EIGEN_DEBUG_ALIGNED_LOAD
3231 return vec_xl(0,
const_cast<double*
>(from));
3235EIGEN_ALWAYS_INLINE Packet2d pload_partial<Packet2d>(
const double* from,
const Index n,
const Index offset) {
3236 return pload_partial_common<Packet2d>(from, n, offset);
3240EIGEN_STRONG_INLINE
void pstore<double>(
double* to,
const Packet2d& from) {
3241 EIGEN_DEBUG_ALIGNED_STORE
3242 vec_xst(from, 0, to);
3246EIGEN_ALWAYS_INLINE
void pstore_partial<double>(
double* to,
const Packet2d& from,
const Index n,
const Index offset) {
3247 pstore_partial_common<Packet2d>(to, from, n, offset);
3251EIGEN_STRONG_INLINE Packet2d pset1<Packet2d>(
const double& from) {
3252 Packet2d v = {from, from};
3256EIGEN_STRONG_INLINE Packet2l pset1<Packet2l>(
const int64_t& from) {
3257 Packet2l v = {from, from};
3262EIGEN_STRONG_INLINE Packet2d pset1frombits<Packet2d>(
unsigned long from) {
3263 Packet2l v = {
static_cast<long long>(from),
static_cast<long long>(from)};
3264 return reinterpret_cast<Packet2d
>(v);
3268EIGEN_STRONG_INLINE
void pbroadcast4<Packet2d>(
const double* a, Packet2d& a0, Packet2d& a1, Packet2d& a2,
3271 a0 = pset1<Packet2d>(a[0]);
3272 a1 = pset1<Packet2d>(a[1]);
3273 a2 = pset1<Packet2d>(a[2]);
3274 a3 = pset1<Packet2d>(a[3]);
3278EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet2d pgather<double, Packet2d>(
const double* from, Index stride) {
3279 return pgather_common<Packet2d>(from, stride);
3282EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet2d pgather_partial<double, Packet2d>(
const double* from, Index stride,
3284 return pgather_common<Packet2d>(from, stride, n);
3287EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE
void pscatter<double, Packet2d>(
double* to,
const Packet2d& from, Index stride) {
3288 pscatter_common<Packet2d>(to, from, stride);
3291EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE
void pscatter_partial<double, Packet2d>(
double* to,
const Packet2d& from,
3292 Index stride,
const Index n) {
3293 pscatter_common<Packet2d>(to, from, stride, n);
3297EIGEN_STRONG_INLINE Packet2d plset<Packet2d>(
const double& a) {
3298 return pset1<Packet2d>(a) + p2d_COUNTDOWN;
3302EIGEN_STRONG_INLINE Packet2d padd<Packet2d>(
const Packet2d& a,
const Packet2d& b) {
3307EIGEN_STRONG_INLINE Packet2d psub<Packet2d>(
const Packet2d& a,
const Packet2d& b) {
3312EIGEN_STRONG_INLINE Packet2d pnegate(
const Packet2d& a) {
3313#ifdef EIGEN_VECTORIZE_POWER8_VECTOR
3316 return vec_xor(a, p2d_MZERO);
3321EIGEN_STRONG_INLINE Packet2d pmul<Packet2d>(
const Packet2d& a,
const Packet2d& b) {
3322 return vec_madd(a, b, p2d_MZERO);
3325EIGEN_STRONG_INLINE Packet2d pdiv<Packet2d>(
const Packet2d& a,
const Packet2d& b) {
3326 return vec_div(a, b);
3331EIGEN_STRONG_INLINE Packet2d pmadd(
const Packet2d& a,
const Packet2d& b,
const Packet2d& c) {
3332 return vec_madd(a, b, c);
3335EIGEN_STRONG_INLINE Packet2d pmsub(
const Packet2d& a,
const Packet2d& b,
const Packet2d& c) {
3336 return vec_msub(a, b, c);
3339EIGEN_STRONG_INLINE Packet2d pnmadd(
const Packet2d& a,
const Packet2d& b,
const Packet2d& c) {
3340 return vec_nmsub(a, b, c);
3343EIGEN_STRONG_INLINE Packet2d pnmsub(
const Packet2d& a,
const Packet2d& b,
const Packet2d& c) {
3344 return vec_nmadd(a, b, c);
3348EIGEN_STRONG_INLINE Packet2d pmin<Packet2d>(
const Packet2d& a,
const Packet2d& b) {
3351 __asm__(
"xvcmpgedp %x0,%x1,%x2\n\txxsel %x0,%x1,%x2,%x0" :
"=&wa"(ret) :
"wa"(a),
"wa"(b));
3356EIGEN_STRONG_INLINE Packet2d pmax<Packet2d>(
const Packet2d& a,
const Packet2d& b) {
3359 __asm__(
"xvcmpgtdp %x0,%x2,%x1\n\txxsel %x0,%x1,%x2,%x0" :
"=&wa"(ret) :
"wa"(a),
"wa"(b));
3364EIGEN_STRONG_INLINE Packet2d pcmp_le(
const Packet2d& a,
const Packet2d& b) {
3365 return reinterpret_cast<Packet2d
>(vec_cmple(a, b));
3368EIGEN_STRONG_INLINE Packet2d pcmp_lt(
const Packet2d& a,
const Packet2d& b) {
3369 return reinterpret_cast<Packet2d
>(vec_cmplt(a, b));
3372EIGEN_STRONG_INLINE Packet2d pcmp_eq(
const Packet2d& a,
const Packet2d& b) {
3373 return reinterpret_cast<Packet2d
>(vec_cmpeq(a, b));
3376#ifdef EIGEN_VECTORIZE_POWER8_VECTOR
3377EIGEN_STRONG_INLINE Packet2l pcmp_eq(
const Packet2l& a,
const Packet2l& b) {
3378 return reinterpret_cast<Packet2l
>(vec_cmpeq(a, b));
3381EIGEN_STRONG_INLINE Packet2l pcmp_eq(
const Packet2l& a,
const Packet2l& b) {
3382 Packet4i halves =
reinterpret_cast<Packet4i
>(vec_cmpeq(
reinterpret_cast<Packet4i
>(a),
reinterpret_cast<Packet4i
>(b)));
3383 Packet4i flipped = vec_perm(halves, halves, p16uc_COMPLEX32_REV);
3384 return reinterpret_cast<Packet2l
>(pand(halves, flipped));
3390#ifdef EIGEN_VECTORIZE_POWER8_VECTOR
3391EIGEN_STRONG_INLINE Packet2l pcmp_lt(
const Packet2l& a,
const Packet2l& b) {
3392 return reinterpret_cast<Packet2l
>(vec_cmplt(a, b));
3395EIGEN_STRONG_INLINE Packet2l pcmp_lt(
const Packet2l& a,
const Packet2l& b) {
3396 const Packet2l ret = {a[0] < b[0] ? -1 : 0, a[1] < b[1] ? -1 : 0};
3401EIGEN_STRONG_INLINE Packet2d pcmp_lt_or_nan(
const Packet2d& a,
const Packet2d& b) {
3402 Packet2d c =
reinterpret_cast<Packet2d
>(vec_cmpge(a, b));
3403 return vec_nor(c, c);
3407EIGEN_STRONG_INLINE Packet2d pand<Packet2d>(
const Packet2d& a,
const Packet2d& b) {
3408 return vec_and(a, b);
3412EIGEN_STRONG_INLINE Packet2d por<Packet2d>(
const Packet2d& a,
const Packet2d& b) {
3413 return vec_or(a, b);
3417EIGEN_STRONG_INLINE Packet2d pxor<Packet2d>(
const Packet2d& a,
const Packet2d& b) {
3418 return vec_xor(a, b);
3422EIGEN_STRONG_INLINE Packet2d pandnot<Packet2d>(
const Packet2d& a,
const Packet2d& b) {
3423 return vec_and(a, vec_nor(b, b));
3427EIGEN_STRONG_INLINE Packet2d pround<Packet2d>(
const Packet2d& a) {
3428 Packet2d t = vec_add(
3429 reinterpret_cast<Packet2d
>(vec_or(vec_and(
reinterpret_cast<Packet2ul
>(a), p2ul_SIGN), p2ul_PREV0DOT5)), a);
3432 __asm__(
"xvrdpiz %x0, %x1\n\t" :
"=&wa"(res) :
"wa"(t));
3437EIGEN_STRONG_INLINE Packet2d pceil<Packet2d>(
const Packet2d& a) {
3441EIGEN_STRONG_INLINE Packet2d pfloor<Packet2d>(
const Packet2d& a) {
3442 return vec_floor(a);
3445EIGEN_STRONG_INLINE Packet2d ptrunc<Packet2d>(
const Packet2d& a) {
3446 return vec_trunc(a);
3449EIGEN_STRONG_INLINE Packet2d print<Packet2d>(
const Packet2d& a) {
3452 __asm__(
"xvrdpic %x0, %x1\n\t" :
"=&wa"(res) :
"wa"(a));
3458EIGEN_STRONG_INLINE Packet2d ploadu<Packet2d>(
const double* from) {
3459 EIGEN_DEBUG_UNALIGNED_LOAD
3460 return vec_xl(0,
const_cast<double*
>(from));
3464EIGEN_ALWAYS_INLINE Packet2d ploadu_partial<Packet2d>(
const double* from,
const Index n,
const Index offset) {
3465 return ploadu_partial_common<Packet2d>(from, n, offset);
3469EIGEN_STRONG_INLINE Packet2d ploaddup<Packet2d>(
const double* from) {
3471 if ((std::ptrdiff_t(from) % 16) == 0)
3472 p = pload<Packet2d>(from);
3474 p = ploadu<Packet2d>(from);
3475 return vec_splat_dbl<0>(p);
3479EIGEN_STRONG_INLINE
void pstoreu<double>(
double* to,
const Packet2d& from) {
3480 EIGEN_DEBUG_UNALIGNED_STORE
3481 vec_xst(from, 0, to);
3485EIGEN_ALWAYS_INLINE
void pstoreu_partial<double>(
double* to,
const Packet2d& from,
const Index n,
const Index offset) {
3486 pstoreu_partial_common<Packet2d>(to, from, n, offset);
3490EIGEN_STRONG_INLINE
void prefetch<double>(
const double* addr) {
3491 EIGEN_PPC_PREFETCH(addr);
3495EIGEN_STRONG_INLINE
double pfirst<Packet2d>(
const Packet2d& a) {
3496 EIGEN_ALIGN16
double x[2];
3497 pstore<double>(x, a);
3502EIGEN_STRONG_INLINE Packet2d preverse(
const Packet2d& a) {
3503 return vec_sld(a, a, 8);
3506EIGEN_STRONG_INLINE Packet2d pabs(
const Packet2d& a) {
3509#ifdef EIGEN_VECTORIZE_POWER8_VECTOR
3511EIGEN_STRONG_INLINE Packet2d psignbit(
const Packet2d& a) {
3512 return (Packet2d)vec_sra((Packet2l)a, vec_splats((
unsigned long long)(63)));
3516static Packet16uc p16uc_DUPSIGN = {0, 0, 0, 0, 0, 0, 0, 0, 8, 8, 8, 8, 8, 8, 8, 8};
3518static Packet16uc p16uc_DUPSIGN = {7, 7, 7, 7, 7, 7, 7, 7, 15, 15, 15, 15, 15, 15, 15, 15};
3522EIGEN_STRONG_INLINE Packet2d psignbit(
const Packet2d& a) {
3523 Packet16c tmp = vec_sra(
reinterpret_cast<Packet16c
>(a), vec_splats((
unsigned char)(7)));
3524 return reinterpret_cast<Packet2d
>(vec_perm(tmp, tmp, p16uc_DUPSIGN));
3529inline Packet2l pcast<Packet2d, Packet2l>(
const Packet2d& x);
3532inline Packet2d pcast<Packet2l, Packet2d>(
const Packet2l& x);
3540#ifdef EIGEN_VECTORIZE_POWER8_VECTOR
3543EIGEN_STRONG_INLINE Packet2l plogical_shift_left(
const Packet2l& a) {
3544 const Packet2ul shift = {N, N};
3545 return vec_sl(a, shift);
3549EIGEN_STRONG_INLINE Packet2l plogical_shift_right(
const Packet2l& a) {
3550 const Packet2ul shift = {N, N};
3551 return vec_sr(a, shift);
3558EIGEN_ALWAYS_INLINE Packet4i shift_even_left(
const Packet4i& a) {
3559 static const Packet16uc perm = {0x14, 0x15, 0x16, 0x17, 0x00, 0x01, 0x02, 0x03,
3560 0x1c, 0x1d, 0x1e, 0x1f, 0x08, 0x09, 0x0a, 0x0b};
3562 return vec_perm(p4i_ZERO, a, perm);
3564 return vec_perm(a, p4i_ZERO, perm);
3570EIGEN_ALWAYS_INLINE Packet4i shift_odd_right(
const Packet4i& a) {
3571 static const Packet16uc perm = {0x04, 0x05, 0x06, 0x07, 0x10, 0x11, 0x12, 0x13,
3572 0x0c, 0x0d, 0x0e, 0x0f, 0x18, 0x19, 0x1a, 0x1b};
3574 return vec_perm(p4i_ZERO, a, perm);
3576 return vec_perm(a, p4i_ZERO, perm);
3580template <
int N,
typename EnableIf =
void>
3581struct plogical_shift_left_impl;
3584struct plogical_shift_left_impl<N, std::enable_if_t<(N < 32) && (N >= 0)> > {
3585 static EIGEN_STRONG_INLINE Packet2l run(
const Packet2l& a) {
3586 static const unsigned n =
static_cast<unsigned>(N);
3587 const Packet4ui shift = {n, n, n, n};
3588 const Packet4i ai =
reinterpret_cast<Packet4i
>(a);
3589 static const unsigned m =
static_cast<unsigned>(32 - N);
3590 const Packet4ui shift_right = {m, m, m, m};
3591 const Packet4i out_hi = vec_sl(ai, shift);
3592 const Packet4i out_lo = shift_even_left(vec_sr(ai, shift_right));
3593 return reinterpret_cast<Packet2l
>(por<Packet4i>(out_hi, out_lo));
3598struct plogical_shift_left_impl<N, std::enable_if_t<(N >= 32)> > {
3599 static EIGEN_STRONG_INLINE Packet2l run(
const Packet2l& a) {
3600 static const unsigned m =
static_cast<unsigned>(N - 32);
3601 const Packet4ui shift = {m, m, m, m};
3602 const Packet4i ai =
reinterpret_cast<Packet4i
>(a);
3603 return reinterpret_cast<Packet2l
>(shift_even_left(vec_sl(ai, shift)));
3608EIGEN_STRONG_INLINE Packet2l plogical_shift_left(
const Packet2l& a) {
3609 return plogical_shift_left_impl<N>::run(a);
3612template <
int N,
typename EnableIf =
void>
3613struct plogical_shift_right_impl;
3616struct plogical_shift_right_impl<N, std::enable_if_t<(N < 32) && (N >= 0)> > {
3617 static EIGEN_STRONG_INLINE Packet2l run(
const Packet2l& a) {
3618 static const unsigned n =
static_cast<unsigned>(N);
3619 const Packet4ui shift = {n, n, n, n};
3620 const Packet4i ai =
reinterpret_cast<Packet4i
>(a);
3621 static const unsigned m =
static_cast<unsigned>(32 - N);
3622 const Packet4ui shift_left = {m, m, m, m};
3623 const Packet4i out_lo = vec_sr(ai, shift);
3624 const Packet4i out_hi = shift_odd_right(vec_sl(ai, shift_left));
3625 return reinterpret_cast<Packet2l
>(por<Packet4i>(out_hi, out_lo));
3630struct plogical_shift_right_impl<N, std::enable_if_t<(N >= 32)> > {
3631 static EIGEN_STRONG_INLINE Packet2l run(
const Packet2l& a) {
3632 static const unsigned m =
static_cast<unsigned>(N - 32);
3633 const Packet4ui shift = {m, m, m, m};
3634 const Packet4i ai =
reinterpret_cast<Packet4i
>(a);
3635 return reinterpret_cast<Packet2l
>(shift_odd_right(vec_sr(ai, shift)));
3640EIGEN_STRONG_INLINE Packet2l plogical_shift_right(
const Packet2l& a) {
3641 return plogical_shift_right_impl<N>::run(a);
3646EIGEN_STRONG_INLINE Packet2d pldexp<Packet2d>(
const Packet2d& a,
const Packet2d& exponent) {
3648 const Packet2d max_exponent = pset1<Packet2d>(2099.0);
3649 const Packet2d last_max = pset1<Packet2d>(1022.0);
3650 const Packet2l e = pcast<Packet2d, Packet2l>(pmin(pmax(exponent, pnegate(max_exponent)), max_exponent));
3651 const Packet2l b = pcast<Packet2d, Packet2l>(pmin(pmax(exponent, pnegate(last_max)), last_max));
3652 const Packet2l bias = {1023, 1023};
3653 const Packet2l even = {-2, -2};
3654 const Packet2l t = psub(e, b) & even;
3655 const Packet2d c1 =
reinterpret_cast<Packet2d
>(plogical_shift_left<51>(t + bias + bias));
3656 const Packet2d c2 =
reinterpret_cast<Packet2d
>(plogical_shift_left<52>(psub(e, t) + bias));
3657 return pldexp_apply_factors(a, c1, c2);
3662EIGEN_STRONG_INLINE Packet2d pfrexp_generic_get_biased_exponent(
const Packet2d& a) {
3663 return pcast<Packet2l, Packet2d>(plogical_shift_right<52>(
reinterpret_cast<Packet2l
>(pabs(a))));
3667EIGEN_STRONG_INLINE Packet2d pfrexp<Packet2d>(
const Packet2d& a, Packet2d& exponent) {
3668 return pfrexp_generic(a, exponent);
3672EIGEN_STRONG_INLINE
double predux<Packet2d>(
const Packet2d& a) {
3674 b =
reinterpret_cast<Packet2d
>(vec_sld(
reinterpret_cast<Packet4f
>(a),
reinterpret_cast<Packet4f
>(a), 8));
3676 return pfirst<Packet2d>(sum);
3680EIGEN_STRONG_INLINE
bool predux_any(
const Packet2d& a) {
3682 const Packet4ui zero = {0, 0, 0, 0};
3683 return vec_any_ne(
reinterpret_cast<Packet4ui
>(a), zero);
3689EIGEN_STRONG_INLINE
double predux_mul<Packet2d>(
const Packet2d& a) {
3691 pmul(a,
reinterpret_cast<Packet2d
>(vec_sld(
reinterpret_cast<Packet4ui
>(a),
reinterpret_cast<Packet4ui
>(a), 8))));
3696EIGEN_STRONG_INLINE
double predux_min<Packet2d>(
const Packet2d& a) {
3698 pmin(a,
reinterpret_cast<Packet2d
>(vec_sld(
reinterpret_cast<Packet4ui
>(a),
reinterpret_cast<Packet4ui
>(a), 8))));
3703EIGEN_STRONG_INLINE
double predux_max<Packet2d>(
const Packet2d& a) {
3705 pmax(a,
reinterpret_cast<Packet2d
>(vec_sld(
reinterpret_cast<Packet4ui
>(a),
reinterpret_cast<Packet4ui
>(a), 8))));
3708EIGEN_DEVICE_FUNC
inline void ptranspose(PacketBlock<Packet2d, 2>& kernel) {
3710 t0 = vec_mergeh(kernel.packet[0], kernel.packet[1]);
3711 t1 = vec_mergel(kernel.packet[0], kernel.packet[1]);
3712 kernel.packet[0] = t0;
3713 kernel.packet[1] = t1;
@ Aligned16
Definition Constants.h:238