Eigen  5.0.1
 
Loading...
Searching...
No Matches
PacketMath.h
1// This file is part of Eigen, a lightweight C++ template library
2// for linear algebra.
3//
4// Copyright (C) 2008-2016 Konstantinos Margaritis <markos@freevec.org>
5//
6// This Source Code Form is subject to the terms of the Mozilla
7// Public License v. 2.0. If a copy of the MPL was not distributed
8// with this file, You can obtain one at http://mozilla.org/MPL/2.0/.
9// SPDX-License-Identifier: MPL-2.0
10
11#ifndef EIGEN_PACKET_MATH_ALTIVEC_H
12#define EIGEN_PACKET_MATH_ALTIVEC_H
13
14// IWYU pragma: private
15#include "../../InternalHeaderCheck.h"
16
17namespace Eigen {
18
19namespace internal {
20
21#ifndef EIGEN_CACHEFRIENDLY_PRODUCT_THRESHOLD
22#define EIGEN_CACHEFRIENDLY_PRODUCT_THRESHOLD 4
23#endif
24
25#ifndef EIGEN_HAS_SINGLE_INSTRUCTION_MADD
26#define EIGEN_HAS_SINGLE_INSTRUCTION_MADD
27#endif
28
29// NOTE Altivec has 32 registers, but Eigen only accepts a value of 8 or 16
30#ifndef EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS
31#define EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS 32
32#endif
33
34typedef __vector float Packet4f;
35typedef __vector int Packet4i;
36typedef __vector unsigned int Packet4ui;
37typedef __vector __bool int Packet4bi;
38typedef __vector short int Packet8s;
39typedef __vector unsigned short int Packet8us;
40typedef __vector __bool short Packet8bi;
41typedef __vector signed char Packet16c;
42typedef __vector unsigned char Packet16uc;
43typedef eigen_packet_wrapper<__vector unsigned short int, 0> Packet8bf;
44
45// To avoid repeating the same code, but we need to reuse the constants
46// and it doesn't really work to declare them global, so we define macros instead
47#define EIGEN_DECLARE_CONST_FAST_Packet4f(NAME, X) Packet4f p4f_##NAME = {X, X, X, X}
48
49#define EIGEN_DECLARE_CONST_FAST_Packet4i(NAME, X) Packet4i p4i_##NAME = vec_splat_s32(X)
50
51#define EIGEN_DECLARE_CONST_FAST_Packet4ui(NAME, X) Packet4ui p4ui_##NAME = {X, X, X, X}
52
53#define EIGEN_DECLARE_CONST_FAST_Packet8us(NAME, X) Packet8us p8us_##NAME = {X, X, X, X, X, X, X, X}
54
55#define EIGEN_DECLARE_CONST_FAST_Packet16uc(NAME, X) \
56 Packet16uc p16uc_##NAME = {X, X, X, X, X, X, X, X, X, X, X, X, X, X, X, X}
57
58#define EIGEN_DECLARE_CONST_Packet4f(NAME, X) Packet4f p4f_##NAME = pset1<Packet4f>(X)
59
60#define EIGEN_DECLARE_CONST_Packet4i(NAME, X) Packet4i p4i_##NAME = pset1<Packet4i>(X)
61
62#define EIGEN_DECLARE_CONST_Packet2d(NAME, X) Packet2d p2d_##NAME = pset1<Packet2d>(X)
63
64#define EIGEN_DECLARE_CONST_Packet2l(NAME, X) Packet2l p2l_##NAME = pset1<Packet2l>(X)
65
66#define EIGEN_DECLARE_CONST_Packet4f_FROM_INT(NAME, X) \
67 const Packet4f p4f_##NAME = reinterpret_cast<Packet4f>(pset1<Packet4i>(X))
68
69#define DST_CHAN 1
70#define DST_CTRL(size, count, stride) (((size) << 24) | ((count) << 16) | (stride))
71#define __UNPACK_TYPE__(PACKETNAME) typename unpacket_traits<PACKETNAME>::type
72
73// These constants are endian-agnostic
74static EIGEN_DECLARE_CONST_FAST_Packet4f(ZERO, 0); //{ 0.0, 0.0, 0.0, 0.0}
75static EIGEN_DECLARE_CONST_FAST_Packet4i(ZERO, 0); //{ 0, 0, 0, 0,}
76static EIGEN_DECLARE_CONST_FAST_Packet4i(ONE, 1); //{ 1, 1, 1, 1}
77static EIGEN_DECLARE_CONST_FAST_Packet4i(MINUS16, -16); //{ -16, -16, -16, -16}
78static EIGEN_DECLARE_CONST_FAST_Packet4i(MINUS1, -1); //{ -1, -1, -1, -1}
79static EIGEN_DECLARE_CONST_FAST_Packet4ui(SIGN, 0x80000000u);
80static EIGEN_DECLARE_CONST_FAST_Packet4ui(PREV0DOT5, 0x3EFFFFFFu);
81static EIGEN_DECLARE_CONST_FAST_Packet8us(ONE, 1); //{ 1, 1, 1, 1, 1, 1, 1, 1}
82static Packet4f p4f_MZERO =
83 (Packet4f)vec_sl((Packet4ui)p4i_MINUS1, (Packet4ui)p4i_MINUS1); //{ 0x80000000, 0x80000000, 0x80000000, 0x80000000}
84#ifndef __VSX__
85static Packet4f p4f_ONE = vec_ctf(p4i_ONE, 0); //{ 1.0, 1.0, 1.0, 1.0}
86#endif
87
88static Packet4f p4f_COUNTDOWN = {0.0, 1.0, 2.0, 3.0};
89static Packet4i p4i_COUNTDOWN = {0, 1, 2, 3};
90static Packet8s p8s_COUNTDOWN = {0, 1, 2, 3, 4, 5, 6, 7};
91static Packet8us p8us_COUNTDOWN = {0, 1, 2, 3, 4, 5, 6, 7};
92
93static Packet16c p16c_COUNTDOWN = {0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15};
94static Packet16uc p16uc_COUNTDOWN = {0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15};
95
96static Packet16uc p16uc_REVERSE32 = {12, 13, 14, 15, 8, 9, 10, 11, 4, 5, 6, 7, 0, 1, 2, 3};
97static Packet16uc p16uc_REVERSE16 = {14, 15, 12, 13, 10, 11, 8, 9, 6, 7, 4, 5, 2, 3, 0, 1};
98static Packet16uc p16uc_REVERSE8 = {15, 14, 13, 12, 11, 10, 9, 8, 7, 6, 5, 4, 3, 2, 1, 0};
99
100#ifdef _BIG_ENDIAN
101static Packet16uc p16uc_DUPLICATE32_HI = {0, 1, 2, 3, 0, 1, 2, 3, 4, 5, 6, 7, 4, 5, 6, 7};
102#endif
103static const Packet16uc p16uc_DUPLICATE16_EVEN = {0, 1, 0, 1, 4, 5, 4, 5, 8, 9, 8, 9, 12, 13, 12, 13};
104static const Packet16uc p16uc_DUPLICATE16_ODD = {2, 3, 2, 3, 6, 7, 6, 7, 10, 11, 10, 11, 14, 15, 14, 15};
105
106static Packet16uc p16uc_QUADRUPLICATE16_HI = {0, 1, 0, 1, 0, 1, 0, 1, 2, 3, 2, 3, 2, 3, 2, 3};
107static Packet16uc p16uc_QUADRUPLICATE16 = {0, 0, 0, 0, 1, 1, 1, 1, 2, 2, 2, 2, 3, 3, 3, 3};
108
109static Packet16uc p16uc_MERGEE16 = {0, 1, 16, 17, 4, 5, 20, 21, 8, 9, 24, 25, 12, 13, 28, 29};
110static Packet16uc p16uc_MERGEO16 = {2, 3, 18, 19, 6, 7, 22, 23, 10, 11, 26, 27, 14, 15, 30, 31};
111#ifdef _BIG_ENDIAN
112static Packet16uc p16uc_MERGEH16 = {0, 1, 4, 5, 8, 9, 12, 13, 16, 17, 20, 21, 24, 25, 28, 29};
113#else
114static Packet16uc p16uc_MERGEL16 = {2, 3, 6, 7, 10, 11, 14, 15, 18, 19, 22, 23, 26, 27, 30, 31};
115#endif
116
117// Handle endianness properly while loading constants
118// Define global static constants:
119#ifdef _BIG_ENDIAN
120static Packet16uc p16uc_FORWARD = vec_lvsl(0, (float*)0);
121static Packet16uc p16uc_PSET32_WODD =
122 vec_sld((Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 0), (Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 2),
123 8); //{ 0,1,2,3, 0,1,2,3, 8,9,10,11, 8,9,10,11 };
124static Packet16uc p16uc_PSET32_WEVEN = vec_sld(p16uc_DUPLICATE32_HI, (Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 3),
125 8); //{ 4,5,6,7, 4,5,6,7, 12,13,14,15, 12,13,14,15 };
126static Packet16uc p16uc_HALF64_0_16 = vec_sld((Packet16uc)p4i_ZERO, vec_splat((Packet16uc)vec_abs(p4i_MINUS16), 3),
127 8); //{ 0,0,0,0, 0,0,0,0, 16,16,16,16, 16,16,16,16};
128#else
129static Packet16uc p16uc_FORWARD = p16uc_REVERSE32;
130static Packet16uc p16uc_PSET32_WODD =
131 vec_sld((Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 1), (Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 3),
132 8); //{ 0,1,2,3, 0,1,2,3, 8,9,10,11, 8,9,10,11 };
133static Packet16uc p16uc_PSET32_WEVEN =
134 vec_sld((Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 0), (Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 2),
135 8); //{ 4,5,6,7, 4,5,6,7, 12,13,14,15, 12,13,14,15 };
136static Packet16uc p16uc_HALF64_0_16 = vec_sld(vec_splat((Packet16uc)vec_abs(p4i_MINUS16), 0), (Packet16uc)p4i_ZERO,
137 8); //{ 0,0,0,0, 0,0,0,0, 16,16,16,16, 16,16,16,16};
138#endif // _BIG_ENDIAN
139
140static Packet16uc p16uc_PSET64_HI = (Packet16uc)vec_mergeh(
141 (Packet4ui)p16uc_PSET32_WODD, (Packet4ui)p16uc_PSET32_WEVEN); //{ 0,1,2,3, 4,5,6,7, 0,1,2,3, 4,5,6,7 };
142static Packet16uc p16uc_PSET64_LO = (Packet16uc)vec_mergel(
143 (Packet4ui)p16uc_PSET32_WODD, (Packet4ui)p16uc_PSET32_WEVEN); //{ 8,9,10,11, 12,13,14,15, 8,9,10,11, 12,13,14,15 };
144static Packet16uc p16uc_TRANSPOSE64_HI =
145 p16uc_PSET64_HI + p16uc_HALF64_0_16; //{ 0,1,2,3, 4,5,6,7, 16,17,18,19, 20,21,22,23};
146static Packet16uc p16uc_TRANSPOSE64_LO =
147 p16uc_PSET64_LO + p16uc_HALF64_0_16; //{ 8,9,10,11, 12,13,14,15, 24,25,26,27, 28,29,30,31};
148
149static Packet16uc p16uc_COMPLEX32_REV =
150 vec_sld(p16uc_REVERSE32, p16uc_REVERSE32, 8); //{ 4,5,6,7, 0,1,2,3, 12,13,14,15, 8,9,10,11 };
151
152#if EIGEN_HAS_BUILTIN(__builtin_prefetch) || EIGEN_COMP_GNUC
153#define EIGEN_PPC_PREFETCH(ADDR) __builtin_prefetch(ADDR);
154#else
155#define EIGEN_PPC_PREFETCH(ADDR) asm(" dcbt [%[addr]]\n" ::[addr] "r"(ADDR) : "cc");
156#endif
157
158#if EIGEN_COMP_LLVM
159#define LOAD_STORE_UNROLL_16 _Pragma("unroll 16")
160#else
161#define LOAD_STORE_UNROLL_16 _Pragma("GCC unroll(16)")
162#endif
163
164template <>
165struct packet_traits<float> : default_packet_traits {
166 typedef Packet4f type;
167 typedef Packet4f half;
168 enum {
169 Vectorizable = 1,
170 AlignedOnScalar = 1,
171 size = 4,
172
173 HasAdd = 1,
174 HasSub = 1,
175 HasMul = 1,
176 HasDiv = 1,
177 HasMin = 1,
178 HasMax = 1,
179 HasAbs = 1,
180 HasSin = EIGEN_FAST_MATH,
181 HasCos = EIGEN_FAST_MATH,
182 HasTan = EIGEN_FAST_MATH,
183 HasACos = 1,
184 HasASin = 1,
185 HasATan = 1,
186 HasATanh = 1,
187 HasLog = 1,
188 HasExp = 1,
189 HasLog1p = 1,
190 HasExpm1 = 1,
191#ifdef EIGEN_VECTORIZE_VSX
192 HasCmp = 1,
193 HasPow = 1,
194 HasSqrt = 1,
195 HasCbrt = 1,
196#if !EIGEN_COMP_CLANG
197 HasRsqrt = 1,
198#else
199 HasRsqrt = 0,
200#endif
201 HasTanh = EIGEN_FAST_MATH,
202 HasErf = EIGEN_FAST_MATH,
203 HasErfc = EIGEN_FAST_MATH,
204#else
205 HasSqrt = 0,
206 HasRsqrt = 0,
207 HasTanh = 0,
208 HasErf = 0,
209#endif
210 HasNegate = 1,
211 };
212};
213template <>
214struct packet_traits<bfloat16> : default_packet_traits {
215 typedef Packet8bf type;
216 typedef Packet8bf half;
217 enum {
218 Vectorizable = 1,
219 AlignedOnScalar = 1,
220 size = 8,
221
222 HasAdd = 1,
223 HasSub = 1,
224 HasMul = 1,
225 HasDiv = 1,
226 HasMin = 1,
227 HasMax = 1,
228 HasAbs = 1,
229 HasSin = EIGEN_FAST_MATH,
230 HasCos = EIGEN_FAST_MATH,
231 HasLog = 1,
232 HasExp = 1,
233#ifdef EIGEN_VECTORIZE_VSX
234 HasSqrt = 1,
235#if !EIGEN_COMP_CLANG
236 HasRsqrt = 1,
237#else
238 HasRsqrt = 0,
239#endif
240#else
241 HasSqrt = 0,
242 HasRsqrt = 0,
243#endif
244 HasTanh = 0,
245 HasErf = 0,
246 HasNegate = 1,
247 };
248};
249
250template <>
251struct packet_traits<int> : default_packet_traits {
252 typedef Packet4i type;
253 typedef Packet4i half;
254 enum {
255 Vectorizable = 1,
256 AlignedOnScalar = 1,
257 size = 4,
258
259 HasAdd = 1,
260 HasSub = 1,
261 HasShift = 1,
262 HasMul = 1,
263#if defined(_ARCH_PWR10) && (EIGEN_COMP_LLVM || EIGEN_GNUC_STRICT_AT_LEAST(11, 0, 0))
264 HasDiv = 1,
265#else
266 HasDiv = 0,
267#endif
268 HasCmp = 1
269 };
270};
271
272template <>
273struct packet_traits<short int> : default_packet_traits {
274 typedef Packet8s type;
275 typedef Packet8s half;
276 enum {
277 Vectorizable = 1,
278 AlignedOnScalar = 1,
279 size = 8,
280
281 HasAdd = 1,
282 HasSub = 1,
283 HasMul = 1,
284 HasDiv = 0,
285 HasCmp = 1
286 };
287};
288
289template <>
290struct packet_traits<unsigned short int> : default_packet_traits {
291 typedef Packet8us type;
292 typedef Packet8us half;
293 enum {
294 Vectorizable = 1,
295 AlignedOnScalar = 1,
296 size = 8,
297
298 HasAdd = 1,
299 HasSub = 1,
300 HasMul = 1,
301 HasDiv = 0,
302 HasCmp = 1
303 };
304};
305
306template <>
307struct packet_traits<signed char> : default_packet_traits {
308 typedef Packet16c type;
309 typedef Packet16c half;
310 enum {
311 Vectorizable = 1,
312 AlignedOnScalar = 1,
313 size = 16,
314
315 HasAdd = 1,
316 HasSub = 1,
317 HasMul = 1,
318 HasDiv = 0,
319 HasCmp = 1
320 };
321};
322
323template <>
324struct packet_traits<unsigned char> : default_packet_traits {
325 typedef Packet16uc type;
326 typedef Packet16uc half;
327 enum {
328 Vectorizable = 1,
329 AlignedOnScalar = 1,
330 size = 16,
331
332 HasAdd = 1,
333 HasSub = 1,
334 HasMul = 1,
335 HasDiv = 0,
336 HasCmp = 1
337 };
338};
339
340template <>
341struct unpacket_traits<Packet4f> {
342 typedef float type;
343 typedef Packet4f half;
344 typedef Packet4i integer_packet;
345 enum {
346 size = 4,
347 alignment = Aligned16,
348 vectorizable = true,
349 masked_load_available = false,
350 masked_store_available = false
351 };
352};
353template <>
354struct unpacket_traits<Packet4i> {
355 typedef int type;
356 typedef Packet4i half;
357 enum {
358 size = 4,
359 alignment = Aligned16,
360 vectorizable = true,
361 masked_load_available = false,
362 masked_store_available = false
363 };
364};
365template <>
366struct unpacket_traits<Packet8s> {
367 typedef short int type;
368 typedef Packet8s half;
369 enum {
370 size = 8,
371 alignment = Aligned16,
372 vectorizable = true,
373 masked_load_available = false,
374 masked_store_available = false
375 };
376};
377template <>
378struct unpacket_traits<Packet8us> {
379 typedef unsigned short int type;
380 typedef Packet8us half;
381 enum {
382 size = 8,
383 alignment = Aligned16,
384 vectorizable = true,
385 masked_load_available = false,
386 masked_store_available = false
387 };
388};
389
390template <>
391struct unpacket_traits<Packet16c> {
392 typedef signed char type;
393 typedef Packet16c half;
394 enum {
395 size = 16,
396 alignment = Aligned16,
397 vectorizable = true,
398 masked_load_available = false,
399 masked_store_available = false
400 };
401};
402template <>
403struct unpacket_traits<Packet16uc> {
404 typedef unsigned char type;
405 typedef Packet16uc half;
406 enum {
407 size = 16,
408 alignment = Aligned16,
409 vectorizable = true,
410 masked_load_available = false,
411 masked_store_available = false
412 };
413};
414
415template <>
416struct unpacket_traits<Packet8bf> {
417 typedef bfloat16 type;
418 typedef Packet8bf half;
419 enum {
420 size = 8,
421 alignment = Aligned16,
422 vectorizable = true,
423 masked_load_available = false,
424 masked_store_available = false
425 };
426};
427
428template <typename Packet>
429EIGEN_STRONG_INLINE Packet pload_common(const __UNPACK_TYPE__(Packet) * from) {
430 // some versions of GCC throw "unused-but-set-parameter".
431 // ignoring these warnings for now.
432 EIGEN_UNUSED_VARIABLE(from);
433 EIGEN_DEBUG_ALIGNED_LOAD
434#ifdef EIGEN_VECTORIZE_VSX
435 return vec_xl(0, const_cast<__UNPACK_TYPE__(Packet)*>(from));
436#else
437 return vec_ld(0, from);
438#endif
439}
440
441// Need to define them first or we get specialization after instantiation errors
442template <>
443EIGEN_STRONG_INLINE Packet4f pload<Packet4f>(const float* from) {
444 return pload_common<Packet4f>(from);
445}
446
447template <>
448EIGEN_STRONG_INLINE Packet4i pload<Packet4i>(const int* from) {
449 return pload_common<Packet4i>(from);
450}
451
452template <>
453EIGEN_STRONG_INLINE Packet8s pload<Packet8s>(const short int* from) {
454 return pload_common<Packet8s>(from);
455}
456
457template <>
458EIGEN_STRONG_INLINE Packet8us pload<Packet8us>(const unsigned short int* from) {
459 return pload_common<Packet8us>(from);
460}
461
462template <>
463EIGEN_STRONG_INLINE Packet16c pload<Packet16c>(const signed char* from) {
464 return pload_common<Packet16c>(from);
465}
466
467template <>
468EIGEN_STRONG_INLINE Packet16uc pload<Packet16uc>(const unsigned char* from) {
469 return pload_common<Packet16uc>(from);
470}
471
472template <>
473EIGEN_STRONG_INLINE Packet8bf pload<Packet8bf>(const bfloat16* from) {
474 return pload_common<Packet8us>(reinterpret_cast<const unsigned short int*>(from));
475}
476
477template <typename Packet>
478EIGEN_ALWAYS_INLINE Packet pload_ignore(const __UNPACK_TYPE__(Packet) * from) {
479 // some versions of GCC throw "unused-but-set-parameter".
480 // ignoring these warnings for now.
481 EIGEN_UNUSED_VARIABLE(from);
482 EIGEN_DEBUG_ALIGNED_LOAD
483 // Ignore partial input memory initialized
484#if !EIGEN_COMP_LLVM
485#pragma GCC diagnostic push
486#pragma GCC diagnostic ignored "-Wmaybe-uninitialized"
487#endif
488#ifdef EIGEN_VECTORIZE_VSX
489 return vec_xl(0, const_cast<__UNPACK_TYPE__(Packet)*>(from));
490#else
491 return vec_ld(0, from);
492#endif
493#if !EIGEN_COMP_LLVM
494#pragma GCC diagnostic pop
495#endif
496}
497
498template <>
499EIGEN_ALWAYS_INLINE Packet8bf pload_ignore<Packet8bf>(const bfloat16* from) {
500 return pload_ignore<Packet8us>(reinterpret_cast<const unsigned short int*>(from));
501}
502
503template <typename Packet>
504EIGEN_ALWAYS_INLINE Packet pload_partial_common(const __UNPACK_TYPE__(Packet) * from, const Index n,
505 const Index offset) {
506 // some versions of GCC throw "unused-but-set-parameter".
507 // ignoring these warnings for now.
508 const Index packet_size = unpacket_traits<Packet>::size;
509 eigen_internal_assert(n + offset <= packet_size && "number of elements plus offset will read past end of packet");
510 const Index size = sizeof(__UNPACK_TYPE__(Packet));
511#ifdef _ARCH_PWR9
512 EIGEN_UNUSED_VARIABLE(packet_size);
513 EIGEN_DEBUG_ALIGNED_LOAD
514 EIGEN_UNUSED_VARIABLE(from);
515 Packet load = vec_xl_len(const_cast<__UNPACK_TYPE__(Packet)*>(from), n * size);
516 if (offset) {
517 Packet16uc shift = pset1<Packet16uc>(offset * 8 * size);
518#ifdef _BIG_ENDIAN
519 load = Packet(vec_sro(Packet16uc(load), shift));
520#else
521 load = Packet(vec_slo(Packet16uc(load), shift));
522#endif
523 }
524 return load;
525#else
526 if (n) {
527 EIGEN_ALIGN16 __UNPACK_TYPE__(Packet) load[packet_size];
528 unsigned char* load2 = reinterpret_cast<unsigned char*>(load + offset);
529 unsigned char* from2 = reinterpret_cast<unsigned char*>(const_cast<__UNPACK_TYPE__(Packet)*>(from));
530 Index n2 = n * size;
531 if (16 <= n2) {
532 pstoreu(load2, ploadu<Packet16uc>(from2));
533 } else {
534 memcpy((void*)load2, (void*)from2, n2);
535 }
536 return pload_ignore<Packet>(load);
537 } else {
538 return Packet(pset1<Packet16uc>(0));
539 }
540#endif
541}
542
543template <>
544EIGEN_ALWAYS_INLINE Packet4f pload_partial<Packet4f>(const float* from, const Index n, const Index offset) {
545 return pload_partial_common<Packet4f>(from, n, offset);
546}
547
548template <>
549EIGEN_ALWAYS_INLINE Packet4i pload_partial<Packet4i>(const int* from, const Index n, const Index offset) {
550 return pload_partial_common<Packet4i>(from, n, offset);
551}
552
553template <>
554EIGEN_ALWAYS_INLINE Packet8s pload_partial<Packet8s>(const short int* from, const Index n, const Index offset) {
555 return pload_partial_common<Packet8s>(from, n, offset);
556}
557
558template <>
559EIGEN_ALWAYS_INLINE Packet8us pload_partial<Packet8us>(const unsigned short int* from, const Index n,
560 const Index offset) {
561 return pload_partial_common<Packet8us>(from, n, offset);
562}
563
564template <>
565EIGEN_ALWAYS_INLINE Packet8bf pload_partial<Packet8bf>(const bfloat16* from, const Index n, const Index offset) {
566 return pload_partial_common<Packet8us>(reinterpret_cast<const unsigned short int*>(from), n, offset);
567}
568
569template <>
570EIGEN_ALWAYS_INLINE Packet16c pload_partial<Packet16c>(const signed char* from, const Index n, const Index offset) {
571 return pload_partial_common<Packet16c>(from, n, offset);
572}
573
574template <>
575EIGEN_ALWAYS_INLINE Packet16uc pload_partial<Packet16uc>(const unsigned char* from, const Index n, const Index offset) {
576 return pload_partial_common<Packet16uc>(from, n, offset);
577}
578
579template <typename Packet>
580EIGEN_STRONG_INLINE void pstore_common(__UNPACK_TYPE__(Packet) * to, const Packet& from) {
581 // some versions of GCC throw "unused-but-set-parameter" (float *to).
582 // ignoring these warnings for now.
583 EIGEN_UNUSED_VARIABLE(to);
584 EIGEN_DEBUG_ALIGNED_STORE
585#ifdef EIGEN_VECTORIZE_VSX
586 vec_xst(from, 0, to);
587#else
588 vec_st(from, 0, to);
589#endif
590}
591
592template <>
593EIGEN_STRONG_INLINE void pstore<float>(float* to, const Packet4f& from) {
594 pstore_common<Packet4f>(to, from);
595}
596
597template <>
598EIGEN_STRONG_INLINE void pstore<int>(int* to, const Packet4i& from) {
599 pstore_common<Packet4i>(to, from);
600}
601
602template <>
603EIGEN_STRONG_INLINE void pstore<short int>(short int* to, const Packet8s& from) {
604 pstore_common<Packet8s>(to, from);
605}
606
607template <>
608EIGEN_STRONG_INLINE void pstore<unsigned short int>(unsigned short int* to, const Packet8us& from) {
609 pstore_common<Packet8us>(to, from);
610}
611
612template <>
613EIGEN_STRONG_INLINE void pstore<bfloat16>(bfloat16* to, const Packet8bf& from) {
614 pstore_common<Packet8us>(reinterpret_cast<unsigned short int*>(to), from.m_val);
615}
616
617template <>
618EIGEN_STRONG_INLINE void pstore<signed char>(signed char* to, const Packet16c& from) {
619 pstore_common<Packet16c>(to, from);
620}
621
622template <>
623EIGEN_STRONG_INLINE void pstore<unsigned char>(unsigned char* to, const Packet16uc& from) {
624 pstore_common<Packet16uc>(to, from);
625}
626
627template <typename Packet>
628EIGEN_ALWAYS_INLINE void pstore_partial_common(__UNPACK_TYPE__(Packet) * to, const Packet& from, const Index n,
629 const Index offset) {
630 // some versions of GCC throw "unused-but-set-parameter" (float *to).
631 // ignoring these warnings for now.
632 const Index packet_size = unpacket_traits<Packet>::size;
633 eigen_internal_assert(n + offset <= packet_size && "number of elements plus offset will write past end of packet");
634 const Index size = sizeof(__UNPACK_TYPE__(Packet));
635#ifdef _ARCH_PWR9
636 EIGEN_UNUSED_VARIABLE(packet_size);
637 EIGEN_UNUSED_VARIABLE(to);
638 EIGEN_DEBUG_ALIGNED_STORE
639 Packet store = from;
640 if (offset) {
641 Packet16uc shift = pset1<Packet16uc>(offset * 8 * size);
642#ifdef _BIG_ENDIAN
643 store = Packet(vec_slo(Packet16uc(store), shift));
644#else
645 store = Packet(vec_sro(Packet16uc(store), shift));
646#endif
647 }
648 vec_xst_len(store, to, n * size);
649#else
650 if (n) {
651 EIGEN_ALIGN16 __UNPACK_TYPE__(Packet) store[packet_size];
652 pstore(store, from);
653 unsigned char* store2 = reinterpret_cast<unsigned char*>(store + offset);
654 unsigned char* to2 = reinterpret_cast<unsigned char*>(to);
655 Index n2 = n * size;
656 if (16 <= n2) {
657 pstore(to2, ploadu<Packet16uc>(store2));
658 } else {
659 memcpy((void*)to2, (void*)store2, n2);
660 }
661 }
662#endif
663}
664
665template <>
666EIGEN_ALWAYS_INLINE void pstore_partial<float>(float* to, const Packet4f& from, const Index n, const Index offset) {
667 pstore_partial_common<Packet4f>(to, from, n, offset);
668}
669
670template <>
671EIGEN_ALWAYS_INLINE void pstore_partial<int>(int* to, const Packet4i& from, const Index n, const Index offset) {
672 pstore_partial_common<Packet4i>(to, from, n, offset);
673}
674
675template <>
676EIGEN_ALWAYS_INLINE void pstore_partial<short int>(short int* to, const Packet8s& from, const Index n,
677 const Index offset) {
678 pstore_partial_common<Packet8s>(to, from, n, offset);
679}
680
681template <>
682EIGEN_ALWAYS_INLINE void pstore_partial<unsigned short int>(unsigned short int* to, const Packet8us& from,
683 const Index n, const Index offset) {
684 pstore_partial_common<Packet8us>(to, from, n, offset);
685}
686
687template <>
688EIGEN_ALWAYS_INLINE void pstore_partial<bfloat16>(bfloat16* to, const Packet8bf& from, const Index n,
689 const Index offset) {
690 pstore_partial_common<Packet8us>(reinterpret_cast<unsigned short int*>(to), from.m_val, n, offset);
691}
692
693template <>
694EIGEN_ALWAYS_INLINE void pstore_partial<signed char>(signed char* to, const Packet16c& from, const Index n,
695 const Index offset) {
696 pstore_partial_common<Packet16c>(to, from, n, offset);
697}
698
699template <>
700EIGEN_ALWAYS_INLINE void pstore_partial<unsigned char>(unsigned char* to, const Packet16uc& from, const Index n,
701 const Index offset) {
702 pstore_partial_common<Packet16uc>(to, from, n, offset);
703}
704
705template <typename Packet>
706EIGEN_STRONG_INLINE Packet pset1_size4(const __UNPACK_TYPE__(Packet) & from) {
707 Packet v = {from, from, from, from};
708 return v;
709}
710
711template <typename Packet>
712EIGEN_STRONG_INLINE Packet pset1_size8(const __UNPACK_TYPE__(Packet) & from) {
713 Packet v = {from, from, from, from, from, from, from, from};
714 return v;
715}
716
717template <typename Packet>
718EIGEN_STRONG_INLINE Packet pset1_size16(const __UNPACK_TYPE__(Packet) & from) {
719 Packet v = {from, from, from, from, from, from, from, from, from, from, from, from, from, from, from, from};
720 return v;
721}
722
723template <>
724EIGEN_STRONG_INLINE Packet4f pset1<Packet4f>(const float& from) {
725 return pset1_size4<Packet4f>(from);
726}
727
728template <>
729EIGEN_STRONG_INLINE Packet4i pset1<Packet4i>(const int& from) {
730 return pset1_size4<Packet4i>(from);
731}
732
733template <>
734EIGEN_STRONG_INLINE Packet8s pset1<Packet8s>(const short int& from) {
735 return pset1_size8<Packet8s>(from);
736}
737
738template <>
739EIGEN_STRONG_INLINE Packet8us pset1<Packet8us>(const unsigned short int& from) {
740 return pset1_size8<Packet8us>(from);
741}
742
743template <>
744EIGEN_STRONG_INLINE Packet16c pset1<Packet16c>(const signed char& from) {
745 return pset1_size16<Packet16c>(from);
746}
747
748template <>
749EIGEN_STRONG_INLINE Packet16uc pset1<Packet16uc>(const unsigned char& from) {
750 return pset1_size16<Packet16uc>(from);
751}
752
753template <>
754EIGEN_STRONG_INLINE Packet4f pset1frombits<Packet4f>(unsigned int from) {
755 return reinterpret_cast<Packet4f>(pset1<Packet4i>(from));
756}
757
758template <>
759EIGEN_STRONG_INLINE Packet8bf pset1<Packet8bf>(const bfloat16& from) {
760 return pset1_size8<Packet8us>(reinterpret_cast<const unsigned short int&>(from));
761}
762
763template <typename Packet>
764EIGEN_STRONG_INLINE void pbroadcast4_common(const __UNPACK_TYPE__(Packet) * a, Packet& a0, Packet& a1, Packet& a2,
765 Packet& a3) {
766 a3 = pload<Packet>(a);
767 a0 = vec_splat(a3, 0);
768 a1 = vec_splat(a3, 1);
769 a2 = vec_splat(a3, 2);
770 a3 = vec_splat(a3, 3);
771}
772
773template <>
774EIGEN_STRONG_INLINE void pbroadcast4<Packet4f>(const float* a, Packet4f& a0, Packet4f& a1, Packet4f& a2, Packet4f& a3) {
775 pbroadcast4_common<Packet4f>(a, a0, a1, a2, a3);
776}
777template <>
778EIGEN_STRONG_INLINE void pbroadcast4<Packet4i>(const int* a, Packet4i& a0, Packet4i& a1, Packet4i& a2, Packet4i& a3) {
779 pbroadcast4_common<Packet4i>(a, a0, a1, a2, a3);
780}
781
782template <typename Packet>
783EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet pgather_common(const __UNPACK_TYPE__(Packet) * from, Index stride,
784 const Index n = unpacket_traits<Packet>::size) {
785 EIGEN_ALIGN16 __UNPACK_TYPE__(Packet) a[unpacket_traits<Packet>::size];
786 eigen_internal_assert(n <= unpacket_traits<Packet>::size && "number of elements will gather past end of packet");
787 if (stride == 1) {
788 if (n == unpacket_traits<Packet>::size) {
789 return ploadu<Packet>(from);
790 } else {
791 return ploadu_partial<Packet>(from, n);
792 }
793 } else {
794 LOAD_STORE_UNROLL_16
795 for (Index i = 0; i < n; i++) {
796 a[i] = from[i * stride];
797 }
798 // Leave rest of the array uninitialized
799 return pload_ignore<Packet>(a);
800 }
801}
802
803template <>
804EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet4f pgather<float, Packet4f>(const float* from, Index stride) {
805 return pgather_common<Packet4f>(from, stride);
806}
807
808template <>
809EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet4i pgather<int, Packet4i>(const int* from, Index stride) {
810 return pgather_common<Packet4i>(from, stride);
811}
812
813template <>
814EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet8s pgather<short int, Packet8s>(const short int* from, Index stride) {
815 return pgather_common<Packet8s>(from, stride);
816}
817
818template <>
819EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet8us pgather<unsigned short int, Packet8us>(const unsigned short int* from,
820 Index stride) {
821 return pgather_common<Packet8us>(from, stride);
822}
823
824template <>
825EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet8bf pgather<bfloat16, Packet8bf>(const bfloat16* from, Index stride) {
826 return pgather_common<Packet8bf>(from, stride);
827}
828
829template <>
830EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet16c pgather<signed char, Packet16c>(const signed char* from, Index stride) {
831 return pgather_common<Packet16c>(from, stride);
832}
833
834template <>
835EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet16uc pgather<unsigned char, Packet16uc>(const unsigned char* from,
836 Index stride) {
837 return pgather_common<Packet16uc>(from, stride);
838}
839
840template <>
841EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet4f pgather_partial<float, Packet4f>(const float* from, Index stride,
842 const Index n) {
843 return pgather_common<Packet4f>(from, stride, n);
844}
845
846template <>
847EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet4i pgather_partial<int, Packet4i>(const int* from, Index stride,
848 const Index n) {
849 return pgather_common<Packet4i>(from, stride, n);
850}
851
852template <>
853EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet8s pgather_partial<short int, Packet8s>(const short int* from, Index stride,
854 const Index n) {
855 return pgather_common<Packet8s>(from, stride, n);
856}
857
858template <>
859EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet8us
860pgather_partial<unsigned short int, Packet8us>(const unsigned short int* from, Index stride, const Index n) {
861 return pgather_common<Packet8us>(from, stride, n);
862}
863
864template <>
865EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet8bf pgather_partial<bfloat16, Packet8bf>(const bfloat16* from, Index stride,
866 const Index n) {
867 return pgather_common<Packet8bf>(from, stride, n);
868}
869
870template <>
871EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet16c pgather_partial<signed char, Packet16c>(const signed char* from,
872 Index stride, const Index n) {
873 return pgather_common<Packet16c>(from, stride, n);
874}
875
876template <>
877EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet16uc pgather_partial<unsigned char, Packet16uc>(const unsigned char* from,
878 Index stride,
879 const Index n) {
880 return pgather_common<Packet16uc>(from, stride, n);
881}
882
883template <typename Packet>
884EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void pscatter_common(__UNPACK_TYPE__(Packet) * to, const Packet& from,
885 Index stride,
886 const Index n = unpacket_traits<Packet>::size) {
887 EIGEN_ALIGN16 __UNPACK_TYPE__(Packet) a[unpacket_traits<Packet>::size];
888 eigen_internal_assert(n <= unpacket_traits<Packet>::size && "number of elements will scatter past end of packet");
889 if (stride == 1) {
890 if (n == unpacket_traits<Packet>::size) {
891 return pstoreu(to, from);
892 } else {
893 return pstoreu_partial(to, from, n);
894 }
895 } else {
896 pstore<__UNPACK_TYPE__(Packet)>(a, from);
897 LOAD_STORE_UNROLL_16
898 for (Index i = 0; i < n; i++) {
899 to[i * stride] = a[i];
900 }
901 }
902}
903
904template <>
905EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void pscatter<float, Packet4f>(float* to, const Packet4f& from, Index stride) {
906 pscatter_common<Packet4f>(to, from, stride);
907}
908
909template <>
910EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void pscatter<int, Packet4i>(int* to, const Packet4i& from, Index stride) {
911 pscatter_common<Packet4i>(to, from, stride);
912}
913
914template <>
915EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void pscatter<short int, Packet8s>(short int* to, const Packet8s& from,
916 Index stride) {
917 pscatter_common<Packet8s>(to, from, stride);
918}
919
920template <>
921EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void pscatter<unsigned short int, Packet8us>(unsigned short int* to,
922 const Packet8us& from,
923 Index stride) {
924 pscatter_common<Packet8us>(to, from, stride);
925}
926
927template <>
928EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void pscatter<bfloat16, Packet8bf>(bfloat16* to, const Packet8bf& from,
929 Index stride) {
930 pscatter_common<Packet8bf>(to, from, stride);
931}
932
933template <>
934EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void pscatter<signed char, Packet16c>(signed char* to, const Packet16c& from,
935 Index stride) {
936 pscatter_common<Packet16c>(to, from, stride);
937}
938
939template <>
940EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void pscatter<unsigned char, Packet16uc>(unsigned char* to,
941 const Packet16uc& from, Index stride) {
942 pscatter_common<Packet16uc>(to, from, stride);
943}
944
945template <>
946EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void pscatter_partial<float, Packet4f>(float* to, const Packet4f& from,
947 Index stride, const Index n) {
948 pscatter_common<Packet4f>(to, from, stride, n);
949}
950
951template <>
952EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void pscatter_partial<int, Packet4i>(int* to, const Packet4i& from, Index stride,
953 const Index n) {
954 pscatter_common<Packet4i>(to, from, stride, n);
955}
956
957template <>
958EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void pscatter_partial<short int, Packet8s>(short int* to, const Packet8s& from,
959 Index stride, const Index n) {
960 pscatter_common<Packet8s>(to, from, stride, n);
961}
962
963template <>
964EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void pscatter_partial<unsigned short int, Packet8us>(unsigned short int* to,
965 const Packet8us& from,
966 Index stride,
967 const Index n) {
968 pscatter_common<Packet8us>(to, from, stride, n);
969}
970
971template <>
972EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void pscatter_partial<bfloat16, Packet8bf>(bfloat16* to, const Packet8bf& from,
973 Index stride, const Index n) {
974 pscatter_common<Packet8bf>(to, from, stride, n);
975}
976
977template <>
978EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void pscatter_partial<signed char, Packet16c>(signed char* to,
979 const Packet16c& from, Index stride,
980 const Index n) {
981 pscatter_common<Packet16c>(to, from, stride, n);
982}
983
984template <>
985EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void pscatter_partial<unsigned char, Packet16uc>(unsigned char* to,
986 const Packet16uc& from,
987 Index stride, const Index n) {
988 pscatter_common<Packet16uc>(to, from, stride, n);
989}
990
991template <>
992EIGEN_STRONG_INLINE Packet4f plset<Packet4f>(const float& a) {
993 return pset1<Packet4f>(a) + p4f_COUNTDOWN;
994}
995template <>
996EIGEN_STRONG_INLINE Packet4i plset<Packet4i>(const int& a) {
997 return pset1<Packet4i>(a) + p4i_COUNTDOWN;
998}
999template <>
1000EIGEN_STRONG_INLINE Packet8s plset<Packet8s>(const short int& a) {
1001 return pset1<Packet8s>(a) + p8s_COUNTDOWN;
1002}
1003template <>
1004EIGEN_STRONG_INLINE Packet8us plset<Packet8us>(const unsigned short int& a) {
1005 return pset1<Packet8us>(a) + p8us_COUNTDOWN;
1006}
1007template <>
1008EIGEN_STRONG_INLINE Packet16c plset<Packet16c>(const signed char& a) {
1009 return pset1<Packet16c>(a) + p16c_COUNTDOWN;
1010}
1011template <>
1012EIGEN_STRONG_INLINE Packet16uc plset<Packet16uc>(const unsigned char& a) {
1013 return pset1<Packet16uc>(a) + p16uc_COUNTDOWN;
1014}
1015
1016template <>
1017EIGEN_STRONG_INLINE Packet4f padd<Packet4f>(const Packet4f& a, const Packet4f& b) {
1018 return a + b;
1019}
1020template <>
1021EIGEN_STRONG_INLINE Packet4i padd<Packet4i>(const Packet4i& a, const Packet4i& b) {
1022 return a + b;
1023}
1024template <>
1025EIGEN_STRONG_INLINE Packet4ui padd<Packet4ui>(const Packet4ui& a, const Packet4ui& b) {
1026 return a + b;
1027}
1028template <>
1029EIGEN_STRONG_INLINE Packet8s padd<Packet8s>(const Packet8s& a, const Packet8s& b) {
1030 return a + b;
1031}
1032template <>
1033EIGEN_STRONG_INLINE Packet8us padd<Packet8us>(const Packet8us& a, const Packet8us& b) {
1034 return a + b;
1035}
1036template <>
1037EIGEN_STRONG_INLINE Packet16c padd<Packet16c>(const Packet16c& a, const Packet16c& b) {
1038 return a + b;
1039}
1040template <>
1041EIGEN_STRONG_INLINE Packet16uc padd<Packet16uc>(const Packet16uc& a, const Packet16uc& b) {
1042 return a + b;
1043}
1044
1045template <>
1046EIGEN_STRONG_INLINE Packet4f psub<Packet4f>(const Packet4f& a, const Packet4f& b) {
1047 return a - b;
1048}
1049template <>
1050EIGEN_STRONG_INLINE Packet4i psub<Packet4i>(const Packet4i& a, const Packet4i& b) {
1051 return a - b;
1052}
1053template <>
1054EIGEN_STRONG_INLINE Packet8s psub<Packet8s>(const Packet8s& a, const Packet8s& b) {
1055 return a - b;
1056}
1057template <>
1058EIGEN_STRONG_INLINE Packet8us psub<Packet8us>(const Packet8us& a, const Packet8us& b) {
1059 return a - b;
1060}
1061template <>
1062EIGEN_STRONG_INLINE Packet16c psub<Packet16c>(const Packet16c& a, const Packet16c& b) {
1063 return a - b;
1064}
1065template <>
1066EIGEN_STRONG_INLINE Packet16uc psub<Packet16uc>(const Packet16uc& a, const Packet16uc& b) {
1067 return a - b;
1068}
1069
1070template <>
1071EIGEN_STRONG_INLINE Packet4f pnegate(const Packet4f& a) {
1072#ifdef EIGEN_VECTORIZE_POWER8_VECTOR
1073 return vec_neg(a);
1074#else
1075 return vec_xor(a, p4f_MZERO);
1076#endif
1077}
1078template <>
1079EIGEN_STRONG_INLINE Packet16c pnegate(const Packet16c& a) {
1080#ifdef EIGEN_VECTORIZE_POWER8_VECTOR
1081 return vec_neg(a);
1082#else
1083 return reinterpret_cast<Packet16c>(p4i_ZERO) - a;
1084#endif
1085}
1086template <>
1087EIGEN_STRONG_INLINE Packet8s pnegate(const Packet8s& a) {
1088#ifdef EIGEN_VECTORIZE_POWER8_VECTOR
1089 return vec_neg(a);
1090#else
1091 return reinterpret_cast<Packet8s>(p4i_ZERO) - a;
1092#endif
1093}
1094template <>
1095EIGEN_STRONG_INLINE Packet4i pnegate(const Packet4i& a) {
1096#ifdef EIGEN_VECTORIZE_POWER8_VECTOR
1097 return vec_neg(a);
1098#else
1099 return p4i_ZERO - a;
1100#endif
1101}
1102
1103template <>
1104EIGEN_STRONG_INLINE Packet4f pmul<Packet4f>(const Packet4f& a, const Packet4f& b) {
1105 return vec_madd(a, b, p4f_MZERO);
1106}
1107template <>
1108EIGEN_STRONG_INLINE Packet4i pmul<Packet4i>(const Packet4i& a, const Packet4i& b) {
1109 return a * b;
1110}
1111template <>
1112EIGEN_STRONG_INLINE Packet8s pmul<Packet8s>(const Packet8s& a, const Packet8s& b) {
1113 return vec_mul(a, b);
1114}
1115template <>
1116EIGEN_STRONG_INLINE Packet8us pmul<Packet8us>(const Packet8us& a, const Packet8us& b) {
1117 return vec_mul(a, b);
1118}
1119template <>
1120EIGEN_STRONG_INLINE Packet16c pmul<Packet16c>(const Packet16c& a, const Packet16c& b) {
1121 return vec_mul(a, b);
1122}
1123template <>
1124EIGEN_STRONG_INLINE Packet16uc pmul<Packet16uc>(const Packet16uc& a, const Packet16uc& b) {
1125 return vec_mul(a, b);
1126}
1127
1128template <>
1129EIGEN_STRONG_INLINE Packet4f pdiv<Packet4f>(const Packet4f& a, const Packet4f& b) {
1130#ifndef __VSX__ // VSX actually provides a div instruction
1131 Packet4f t, y_0, y_1;
1132
1133 // Altivec does not offer a divide instruction, we have to do a reciprocal approximation
1134 y_0 = vec_re(b);
1135
1136 // Do one Newton-Raphson iteration to get the needed accuracy
1137 t = vec_nmsub(y_0, b, p4f_ONE);
1138 y_1 = vec_madd(y_0, t, y_0);
1139
1140 return vec_madd(a, y_1, p4f_MZERO);
1141#else
1142 return vec_div(a, b);
1143#endif
1144}
1145
1146template <>
1147EIGEN_STRONG_INLINE Packet4i pdiv<Packet4i>(const Packet4i& a, const Packet4i& b) {
1148#if defined(_ARCH_PWR10) && (EIGEN_COMP_LLVM || EIGEN_GNUC_STRICT_AT_LEAST(11, 0, 0))
1149 return vec_div(a, b);
1150#else
1151 EIGEN_UNUSED_VARIABLE(a);
1152 EIGEN_UNUSED_VARIABLE(b);
1153 eigen_assert(false && "packet integer division are not supported by AltiVec");
1154 return pset1<Packet4i>(0);
1155#endif
1156}
1157
1158// This overload is required for integer packet types.
1159template <>
1160EIGEN_STRONG_INLINE Packet4f pmadd(const Packet4f& a, const Packet4f& b, const Packet4f& c) {
1161 return vec_madd(a, b, c);
1162}
1163template <>
1164EIGEN_STRONG_INLINE Packet4i pmadd(const Packet4i& a, const Packet4i& b, const Packet4i& c) {
1165 return a * b + c;
1166}
1167template <>
1168EIGEN_STRONG_INLINE Packet8s pmadd(const Packet8s& a, const Packet8s& b, const Packet8s& c) {
1169 return vec_madd(a, b, c);
1170}
1171template <>
1172EIGEN_STRONG_INLINE Packet8us pmadd(const Packet8us& a, const Packet8us& b, const Packet8us& c) {
1173 return vec_madd(a, b, c);
1174}
1175
1176template <>
1177EIGEN_STRONG_INLINE Packet4f pmsub(const Packet4f& a, const Packet4f& b, const Packet4f& c) {
1178#ifdef EIGEN_VECTORIZE_VSX
1179 return vec_msub(a, b, c);
1180#else
1181 return vec_madd(a, b, pnegate(c));
1182#endif
1183}
1184
1185#ifdef EIGEN_VECTORIZE_VSX
1186template <>
1187EIGEN_STRONG_INLINE Packet4f pnmadd(const Packet4f& a, const Packet4f& b, const Packet4f& c) {
1188 return vec_nmsub(a, b, c);
1189}
1190template <>
1191EIGEN_STRONG_INLINE Packet4f pnmsub(const Packet4f& a, const Packet4f& b, const Packet4f& c) {
1192 return vec_nmadd(a, b, c);
1193}
1194#endif
1195
1196template <>
1197EIGEN_STRONG_INLINE Packet4f pmin<Packet4f>(const Packet4f& a, const Packet4f& b) {
1198#ifdef EIGEN_VECTORIZE_VSX
1199 // NOTE: about 10% slower than vec_min, but consistent with std::min and SSE regarding NaN
1200 Packet4f ret;
1201 __asm__("xvcmpgesp %x0,%x1,%x2\n\txxsel %x0,%x1,%x2,%x0" : "=&wa"(ret) : "wa"(a), "wa"(b));
1202 return ret;
1203#else
1204 return vec_min(a, b);
1205#endif
1206}
1207template <>
1208EIGEN_STRONG_INLINE Packet4i pmin<Packet4i>(const Packet4i& a, const Packet4i& b) {
1209 return vec_min(a, b);
1210}
1211template <>
1212EIGEN_STRONG_INLINE Packet8s pmin<Packet8s>(const Packet8s& a, const Packet8s& b) {
1213 return vec_min(a, b);
1214}
1215template <>
1216EIGEN_STRONG_INLINE Packet8us pmin<Packet8us>(const Packet8us& a, const Packet8us& b) {
1217 return vec_min(a, b);
1218}
1219template <>
1220EIGEN_STRONG_INLINE Packet16c pmin<Packet16c>(const Packet16c& a, const Packet16c& b) {
1221 return vec_min(a, b);
1222}
1223template <>
1224EIGEN_STRONG_INLINE Packet16uc pmin<Packet16uc>(const Packet16uc& a, const Packet16uc& b) {
1225 return vec_min(a, b);
1226}
1227
1228template <>
1229EIGEN_STRONG_INLINE Packet4f pmax<Packet4f>(const Packet4f& a, const Packet4f& b) {
1230#ifdef EIGEN_VECTORIZE_VSX
1231 // NOTE: about 10% slower than vec_max, but consistent with std::max and SSE regarding NaN
1232 Packet4f ret;
1233 __asm__("xvcmpgtsp %x0,%x2,%x1\n\txxsel %x0,%x1,%x2,%x0" : "=&wa"(ret) : "wa"(a), "wa"(b));
1234 return ret;
1235#else
1236 return vec_max(a, b);
1237#endif
1238}
1239template <>
1240EIGEN_STRONG_INLINE Packet4i pmax<Packet4i>(const Packet4i& a, const Packet4i& b) {
1241 return vec_max(a, b);
1242}
1243template <>
1244EIGEN_STRONG_INLINE Packet8s pmax<Packet8s>(const Packet8s& a, const Packet8s& b) {
1245 return vec_max(a, b);
1246}
1247template <>
1248EIGEN_STRONG_INLINE Packet8us pmax<Packet8us>(const Packet8us& a, const Packet8us& b) {
1249 return vec_max(a, b);
1250}
1251template <>
1252EIGEN_STRONG_INLINE Packet16c pmax<Packet16c>(const Packet16c& a, const Packet16c& b) {
1253 return vec_max(a, b);
1254}
1255template <>
1256EIGEN_STRONG_INLINE Packet16uc pmax<Packet16uc>(const Packet16uc& a, const Packet16uc& b) {
1257 return vec_max(a, b);
1258}
1259
1260template <>
1261EIGEN_STRONG_INLINE Packet4f pcmp_le(const Packet4f& a, const Packet4f& b) {
1262 return reinterpret_cast<Packet4f>(vec_cmple(a, b));
1263}
1264// To fix bug with vec_cmplt on older versions
1265#ifdef EIGEN_VECTORIZE_VSX
1266template <>
1267EIGEN_STRONG_INLINE Packet4f pcmp_lt(const Packet4f& a, const Packet4f& b) {
1268 return reinterpret_cast<Packet4f>(vec_cmplt(a, b));
1269}
1270#endif
1271template <>
1272EIGEN_STRONG_INLINE Packet4f pcmp_eq(const Packet4f& a, const Packet4f& b) {
1273 return reinterpret_cast<Packet4f>(vec_cmpeq(a, b));
1274}
1275template <>
1276EIGEN_STRONG_INLINE Packet4f pcmp_lt_or_nan(const Packet4f& a, const Packet4f& b) {
1277 Packet4f c = reinterpret_cast<Packet4f>(vec_cmpge(a, b));
1278 return vec_nor(c, c);
1279}
1280
1281#ifdef EIGEN_VECTORIZE_VSX
1282template <>
1283EIGEN_STRONG_INLINE Packet4i pcmp_le(const Packet4i& a, const Packet4i& b) {
1284 return reinterpret_cast<Packet4i>(vec_cmple(a, b));
1285}
1286#endif
1287template <>
1288EIGEN_STRONG_INLINE Packet4i pcmp_lt(const Packet4i& a, const Packet4i& b) {
1289 return reinterpret_cast<Packet4i>(vec_cmplt(a, b));
1290}
1291template <>
1292EIGEN_STRONG_INLINE Packet4i pcmp_eq(const Packet4i& a, const Packet4i& b) {
1293 return reinterpret_cast<Packet4i>(vec_cmpeq(a, b));
1294}
1295#ifdef EIGEN_VECTORIZE_VSX
1296template <>
1297EIGEN_STRONG_INLINE Packet8s pcmp_le(const Packet8s& a, const Packet8s& b) {
1298 return reinterpret_cast<Packet8s>(vec_cmple(a, b));
1299}
1300#endif
1301template <>
1302EIGEN_STRONG_INLINE Packet8s pcmp_lt(const Packet8s& a, const Packet8s& b) {
1303 return reinterpret_cast<Packet8s>(vec_cmplt(a, b));
1304}
1305template <>
1306EIGEN_STRONG_INLINE Packet8s pcmp_eq(const Packet8s& a, const Packet8s& b) {
1307 return reinterpret_cast<Packet8s>(vec_cmpeq(a, b));
1308}
1309#ifdef EIGEN_VECTORIZE_VSX
1310template <>
1311EIGEN_STRONG_INLINE Packet8us pcmp_le(const Packet8us& a, const Packet8us& b) {
1312 return reinterpret_cast<Packet8us>(vec_cmple(a, b));
1313}
1314#endif
1315template <>
1316EIGEN_STRONG_INLINE Packet8us pcmp_lt(const Packet8us& a, const Packet8us& b) {
1317 return reinterpret_cast<Packet8us>(vec_cmplt(a, b));
1318}
1319template <>
1320EIGEN_STRONG_INLINE Packet8us pcmp_eq(const Packet8us& a, const Packet8us& b) {
1321 return reinterpret_cast<Packet8us>(vec_cmpeq(a, b));
1322}
1323#ifdef EIGEN_VECTORIZE_VSX
1324template <>
1325EIGEN_STRONG_INLINE Packet16c pcmp_le(const Packet16c& a, const Packet16c& b) {
1326 return reinterpret_cast<Packet16c>(vec_cmple(a, b));
1327}
1328#endif
1329template <>
1330EIGEN_STRONG_INLINE Packet16c pcmp_lt(const Packet16c& a, const Packet16c& b) {
1331 return reinterpret_cast<Packet16c>(vec_cmplt(a, b));
1332}
1333template <>
1334EIGEN_STRONG_INLINE Packet16c pcmp_eq(const Packet16c& a, const Packet16c& b) {
1335 return reinterpret_cast<Packet16c>(vec_cmpeq(a, b));
1336}
1337#ifdef EIGEN_VECTORIZE_VSX
1338template <>
1339EIGEN_STRONG_INLINE Packet16uc pcmp_le(const Packet16uc& a, const Packet16uc& b) {
1340 return reinterpret_cast<Packet16uc>(vec_cmple(a, b));
1341}
1342#endif
1343template <>
1344EIGEN_STRONG_INLINE Packet16uc pcmp_lt(const Packet16uc& a, const Packet16uc& b) {
1345 return reinterpret_cast<Packet16uc>(vec_cmplt(a, b));
1346}
1347template <>
1348EIGEN_STRONG_INLINE Packet16uc pcmp_eq(const Packet16uc& a, const Packet16uc& b) {
1349 return reinterpret_cast<Packet16uc>(vec_cmpeq(a, b));
1350}
1351
1352template <>
1353EIGEN_STRONG_INLINE Packet4f pand<Packet4f>(const Packet4f& a, const Packet4f& b) {
1354 return vec_and(a, b);
1355}
1356template <>
1357EIGEN_STRONG_INLINE Packet4i pand<Packet4i>(const Packet4i& a, const Packet4i& b) {
1358 return vec_and(a, b);
1359}
1360template <>
1361EIGEN_STRONG_INLINE Packet4ui pand<Packet4ui>(const Packet4ui& a, const Packet4ui& b) {
1362 return vec_and(a, b);
1363}
1364template <>
1365EIGEN_STRONG_INLINE Packet8us pand<Packet8us>(const Packet8us& a, const Packet8us& b) {
1366 return vec_and(a, b);
1367}
1368template <>
1369EIGEN_STRONG_INLINE Packet8bf pand<Packet8bf>(const Packet8bf& a, const Packet8bf& b) {
1370 return pand<Packet8us>(a, b);
1371}
1372
1373template <>
1374EIGEN_STRONG_INLINE Packet4f por<Packet4f>(const Packet4f& a, const Packet4f& b) {
1375 return vec_or(a, b);
1376}
1377template <>
1378EIGEN_STRONG_INLINE Packet4i por<Packet4i>(const Packet4i& a, const Packet4i& b) {
1379 return vec_or(a, b);
1380}
1381template <>
1382EIGEN_STRONG_INLINE Packet8s por<Packet8s>(const Packet8s& a, const Packet8s& b) {
1383 return vec_or(a, b);
1384}
1385template <>
1386EIGEN_STRONG_INLINE Packet8us por<Packet8us>(const Packet8us& a, const Packet8us& b) {
1387 return vec_or(a, b);
1388}
1389template <>
1390EIGEN_STRONG_INLINE Packet8bf por<Packet8bf>(const Packet8bf& a, const Packet8bf& b) {
1391 return por<Packet8us>(a, b);
1392}
1393
1394template <>
1395EIGEN_STRONG_INLINE Packet4f pxor<Packet4f>(const Packet4f& a, const Packet4f& b) {
1396 return vec_xor(a, b);
1397}
1398template <>
1399EIGEN_STRONG_INLINE Packet4i pxor<Packet4i>(const Packet4i& a, const Packet4i& b) {
1400 return vec_xor(a, b);
1401}
1402template <>
1403EIGEN_STRONG_INLINE Packet8us pxor<Packet8us>(const Packet8us& a, const Packet8us& b) {
1404 return vec_xor(a, b);
1405}
1406template <>
1407EIGEN_STRONG_INLINE Packet8bf pxor<Packet8bf>(const Packet8bf& a, const Packet8bf& b) {
1408 return pxor<Packet8us>(a, b);
1409}
1410
1411template <>
1412EIGEN_STRONG_INLINE Packet4f pandnot<Packet4f>(const Packet4f& a, const Packet4f& b) {
1413 return vec_andc(a, b);
1414}
1415template <>
1416EIGEN_STRONG_INLINE Packet4i pandnot<Packet4i>(const Packet4i& a, const Packet4i& b) {
1417 return vec_andc(a, b);
1418}
1419
1420template <>
1421EIGEN_STRONG_INLINE Packet4f pselect(const Packet4f& mask, const Packet4f& a, const Packet4f& b) {
1422 return vec_sel(b, a, reinterpret_cast<Packet4ui>(mask));
1423}
1424
1425template <>
1426EIGEN_STRONG_INLINE Packet4f pround<Packet4f>(const Packet4f& a) {
1427 Packet4f t = vec_add(
1428 reinterpret_cast<Packet4f>(vec_or(vec_and(reinterpret_cast<Packet4ui>(a), p4ui_SIGN), p4ui_PREV0DOT5)), a);
1429 Packet4f res;
1430
1431#ifdef EIGEN_VECTORIZE_VSX
1432 __asm__("xvrspiz %x0, %x1\n\t" : "=&wa"(res) : "wa"(t));
1433#else
1434 __asm__("vrfiz %0, %1\n\t" : "=v"(res) : "v"(t));
1435#endif
1436
1437 return res;
1438}
1439template <>
1440EIGEN_STRONG_INLINE Packet4f pceil<Packet4f>(const Packet4f& a) {
1441 return vec_ceil(a);
1442}
1443template <>
1444EIGEN_STRONG_INLINE Packet4f pfloor<Packet4f>(const Packet4f& a) {
1445 return vec_floor(a);
1446}
1447template <>
1448EIGEN_STRONG_INLINE Packet4f ptrunc<Packet4f>(const Packet4f& a) {
1449 return vec_trunc(a);
1450}
1451#ifdef EIGEN_VECTORIZE_VSX
1452template <>
1453EIGEN_STRONG_INLINE Packet4f print<Packet4f>(const Packet4f& a) {
1454 Packet4f res;
1455
1456 __asm__("xvrspic %x0, %x1\n\t" : "=&wa"(res) : "wa"(a));
1457
1458 return res;
1459}
1460#endif
1461
1462template <typename Packet>
1463EIGEN_STRONG_INLINE Packet ploadu_common(const __UNPACK_TYPE__(Packet) * from) {
1464 EIGEN_DEBUG_UNALIGNED_LOAD
1465#if defined(EIGEN_VECTORIZE_VSX)
1466 return vec_xl(0, const_cast<__UNPACK_TYPE__(Packet)*>(from));
1467#else
1468 Packet16uc MSQ = vec_ld(0, (unsigned char*)from); // most significant quadword
1469 Packet16uc LSQ = vec_ld(15, (unsigned char*)from); // least significant quadword
1470 Packet16uc mask = vec_lvsl(0, from); // create the permute mask
1471 // TODO: Add static_cast here
1472 return (Packet)vec_perm(MSQ, LSQ, mask); // align the data
1473#endif
1474}
1475
1476template <>
1477EIGEN_STRONG_INLINE Packet4f ploadu<Packet4f>(const float* from) {
1478 return ploadu_common<Packet4f>(from);
1479}
1480template <>
1481EIGEN_STRONG_INLINE Packet4i ploadu<Packet4i>(const int* from) {
1482 return ploadu_common<Packet4i>(from);
1483}
1484template <>
1485EIGEN_STRONG_INLINE Packet8s ploadu<Packet8s>(const short int* from) {
1486 return ploadu_common<Packet8s>(from);
1487}
1488template <>
1489EIGEN_STRONG_INLINE Packet8us ploadu<Packet8us>(const unsigned short int* from) {
1490 return ploadu_common<Packet8us>(from);
1491}
1492template <>
1493EIGEN_STRONG_INLINE Packet8bf ploadu<Packet8bf>(const bfloat16* from) {
1494 return ploadu_common<Packet8us>(reinterpret_cast<const unsigned short int*>(from));
1495}
1496template <>
1497EIGEN_STRONG_INLINE Packet16c ploadu<Packet16c>(const signed char* from) {
1498 return ploadu_common<Packet16c>(from);
1499}
1500template <>
1501EIGEN_STRONG_INLINE Packet16uc ploadu<Packet16uc>(const unsigned char* from) {
1502 return ploadu_common<Packet16uc>(from);
1503}
1504
1505template <typename Packet>
1506EIGEN_ALWAYS_INLINE Packet ploadu_partial_common(const __UNPACK_TYPE__(Packet) * from, const Index n,
1507 const Index offset) {
1508 const Index packet_size = unpacket_traits<Packet>::size;
1509 eigen_internal_assert(n + offset <= packet_size && "number of elements plus offset will read past end of packet");
1510 const Index size = sizeof(__UNPACK_TYPE__(Packet));
1511#ifdef _ARCH_PWR9
1512 EIGEN_UNUSED_VARIABLE(packet_size);
1513 EIGEN_DEBUG_ALIGNED_LOAD
1514 EIGEN_DEBUG_UNALIGNED_LOAD
1515 Packet load = vec_xl_len(const_cast<__UNPACK_TYPE__(Packet)*>(from), n * size);
1516 if (offset) {
1517 Packet16uc shift = pset1<Packet16uc>(offset * 8 * size);
1518#ifdef _BIG_ENDIAN
1519 load = Packet(vec_sro(Packet16uc(load), shift));
1520#else
1521 load = Packet(vec_slo(Packet16uc(load), shift));
1522#endif
1523 }
1524 return load;
1525#else
1526 if (n) {
1527 EIGEN_ALIGN16 __UNPACK_TYPE__(Packet) load[packet_size];
1528 unsigned char* load2 = reinterpret_cast<unsigned char*>(load + offset);
1529 unsigned char* from2 = reinterpret_cast<unsigned char*>(const_cast<__UNPACK_TYPE__(Packet)*>(from));
1530 Index n2 = n * size;
1531 if (16 <= n2) {
1532 pstoreu(load2, ploadu<Packet16uc>(from2));
1533 } else {
1534 memcpy((void*)load2, (void*)from2, n2);
1535 }
1536 return pload_ignore<Packet>(load);
1537 } else {
1538 return Packet(pset1<Packet16uc>(0));
1539 }
1540#endif
1541}
1542
1543template <>
1544EIGEN_ALWAYS_INLINE Packet4f ploadu_partial<Packet4f>(const float* from, const Index n, const Index offset) {
1545 return ploadu_partial_common<Packet4f>(from, n, offset);
1546}
1547template <>
1548EIGEN_ALWAYS_INLINE Packet4i ploadu_partial<Packet4i>(const int* from, const Index n, const Index offset) {
1549 return ploadu_partial_common<Packet4i>(from, n, offset);
1550}
1551template <>
1552EIGEN_ALWAYS_INLINE Packet8s ploadu_partial<Packet8s>(const short int* from, const Index n, const Index offset) {
1553 return ploadu_partial_common<Packet8s>(from, n, offset);
1554}
1555template <>
1556EIGEN_ALWAYS_INLINE Packet8us ploadu_partial<Packet8us>(const unsigned short int* from, const Index n,
1557 const Index offset) {
1558 return ploadu_partial_common<Packet8us>(from, n, offset);
1559}
1560template <>
1561EIGEN_ALWAYS_INLINE Packet8bf ploadu_partial<Packet8bf>(const bfloat16* from, const Index n, const Index offset) {
1562 return ploadu_partial_common<Packet8us>(reinterpret_cast<const unsigned short int*>(from), n, offset);
1563}
1564template <>
1565EIGEN_ALWAYS_INLINE Packet16c ploadu_partial<Packet16c>(const signed char* from, const Index n, const Index offset) {
1566 return ploadu_partial_common<Packet16c>(from, n, offset);
1567}
1568template <>
1569EIGEN_ALWAYS_INLINE Packet16uc ploadu_partial<Packet16uc>(const unsigned char* from, const Index n,
1570 const Index offset) {
1571 return ploadu_partial_common<Packet16uc>(from, n, offset);
1572}
1573
1574template <typename Packet>
1575EIGEN_STRONG_INLINE Packet ploaddup_common(const __UNPACK_TYPE__(Packet) * from) {
1576 Packet p;
1577 if ((std::ptrdiff_t(from) % 16) == 0)
1578 p = pload<Packet>(from);
1579 else
1580 p = ploadu<Packet>(from);
1581 return vec_mergeh(p, p);
1582}
1583template <>
1584EIGEN_STRONG_INLINE Packet4f ploaddup<Packet4f>(const float* from) {
1585 return ploaddup_common<Packet4f>(from);
1586}
1587template <>
1588EIGEN_STRONG_INLINE Packet4i ploaddup<Packet4i>(const int* from) {
1589 return ploaddup_common<Packet4i>(from);
1590}
1591
1592template <>
1593EIGEN_STRONG_INLINE Packet8s ploaddup<Packet8s>(const short int* from) {
1594 Packet8s p;
1595 if ((std::ptrdiff_t(from) % 16) == 0)
1596 p = pload<Packet8s>(from);
1597 else
1598 p = ploadu<Packet8s>(from);
1599 return vec_mergeh(p, p);
1600}
1601
1602template <>
1603EIGEN_STRONG_INLINE Packet8us ploaddup<Packet8us>(const unsigned short int* from) {
1604 Packet8us p;
1605 if ((std::ptrdiff_t(from) % 16) == 0)
1606 p = pload<Packet8us>(from);
1607 else
1608 p = ploadu<Packet8us>(from);
1609 return vec_mergeh(p, p);
1610}
1611
1612template <>
1613EIGEN_STRONG_INLINE Packet8s ploadquad<Packet8s>(const short int* from) {
1614 Packet8s p;
1615 if ((std::ptrdiff_t(from) % 16) == 0)
1616 p = pload<Packet8s>(from);
1617 else
1618 p = ploadu<Packet8s>(from);
1619 return vec_perm(p, p, p16uc_QUADRUPLICATE16_HI);
1620}
1621
1622template <>
1623EIGEN_STRONG_INLINE Packet8us ploadquad<Packet8us>(const unsigned short int* from) {
1624 Packet8us p;
1625 if ((std::ptrdiff_t(from) % 16) == 0)
1626 p = pload<Packet8us>(from);
1627 else
1628 p = ploadu<Packet8us>(from);
1629 return vec_perm(p, p, p16uc_QUADRUPLICATE16_HI);
1630}
1631
1632template <>
1633EIGEN_STRONG_INLINE Packet8bf ploadquad<Packet8bf>(const bfloat16* from) {
1634 return ploadquad<Packet8us>(reinterpret_cast<const unsigned short int*>(from));
1635}
1636
1637template <>
1638EIGEN_STRONG_INLINE Packet16c ploaddup<Packet16c>(const signed char* from) {
1639 Packet16c p;
1640 if ((std::ptrdiff_t(from) % 16) == 0)
1641 p = pload<Packet16c>(from);
1642 else
1643 p = ploadu<Packet16c>(from);
1644 return vec_mergeh(p, p);
1645}
1646
1647template <>
1648EIGEN_STRONG_INLINE Packet16uc ploaddup<Packet16uc>(const unsigned char* from) {
1649 Packet16uc p;
1650 if ((std::ptrdiff_t(from) % 16) == 0)
1651 p = pload<Packet16uc>(from);
1652 else
1653 p = ploadu<Packet16uc>(from);
1654 return vec_mergeh(p, p);
1655}
1656
1657template <>
1658EIGEN_STRONG_INLINE Packet16c ploadquad<Packet16c>(const signed char* from) {
1659 Packet16c p;
1660 if ((std::ptrdiff_t(from) % 16) == 0)
1661 p = pload<Packet16c>(from);
1662 else
1663 p = ploadu<Packet16c>(from);
1664 return vec_perm(p, p, p16uc_QUADRUPLICATE16);
1665}
1666
1667template <>
1668EIGEN_STRONG_INLINE Packet16uc ploadquad<Packet16uc>(const unsigned char* from) {
1669 Packet16uc p;
1670 if ((std::ptrdiff_t(from) % 16) == 0)
1671 p = pload<Packet16uc>(from);
1672 else
1673 p = ploadu<Packet16uc>(from);
1674 return vec_perm(p, p, p16uc_QUADRUPLICATE16);
1675}
1676
1677template <typename Packet>
1678EIGEN_STRONG_INLINE void pstoreu_common(__UNPACK_TYPE__(Packet) * to, const Packet& from) {
1679 EIGEN_DEBUG_UNALIGNED_STORE
1680#if defined(EIGEN_VECTORIZE_VSX)
1681 vec_xst(from, 0, to);
1682#else
1683 // Taken from http://developer.apple.com/hardwaredrivers/ve/alignment.html
1684 // Warning: not thread safe!
1685 Packet16uc MSQ, LSQ, edges;
1686 Packet16uc edgeAlign, align;
1687
1688 MSQ = vec_ld(0, (unsigned char*)to); // most significant quadword
1689 LSQ = vec_ld(15, (unsigned char*)to); // least significant quadword
1690 edgeAlign = vec_lvsl(0, to); // permute map to extract edges
1691 edges = vec_perm(LSQ, MSQ, edgeAlign); // extract the edges
1692 align = vec_lvsr(0, to); // permute map to misalign data
1693 MSQ = vec_perm(edges, (Packet16uc)from, align); // misalign the data (MSQ)
1694 LSQ = vec_perm((Packet16uc)from, edges, align); // misalign the data (LSQ)
1695 vec_st(LSQ, 15, (unsigned char*)to); // Store the LSQ part first
1696 vec_st(MSQ, 0, (unsigned char*)to); // Store the MSQ part second
1697#endif
1698}
1699template <>
1700EIGEN_STRONG_INLINE void pstoreu<float>(float* to, const Packet4f& from) {
1701 pstoreu_common<Packet4f>(to, from);
1702}
1703template <>
1704EIGEN_STRONG_INLINE void pstoreu<int>(int* to, const Packet4i& from) {
1705 pstoreu_common<Packet4i>(to, from);
1706}
1707template <>
1708EIGEN_STRONG_INLINE void pstoreu<short int>(short int* to, const Packet8s& from) {
1709 pstoreu_common<Packet8s>(to, from);
1710}
1711template <>
1712EIGEN_STRONG_INLINE void pstoreu<unsigned short int>(unsigned short int* to, const Packet8us& from) {
1713 pstoreu_common<Packet8us>(to, from);
1714}
1715template <>
1716EIGEN_STRONG_INLINE void pstoreu<bfloat16>(bfloat16* to, const Packet8bf& from) {
1717 pstoreu_common<Packet8us>(reinterpret_cast<unsigned short int*>(to), from.m_val);
1718}
1719template <>
1720EIGEN_STRONG_INLINE void pstoreu<signed char>(signed char* to, const Packet16c& from) {
1721 pstoreu_common<Packet16c>(to, from);
1722}
1723template <>
1724EIGEN_STRONG_INLINE void pstoreu<unsigned char>(unsigned char* to, const Packet16uc& from) {
1725 pstoreu_common<Packet16uc>(to, from);
1726}
1727
1728template <typename Packet>
1729EIGEN_ALWAYS_INLINE void pstoreu_partial_common(__UNPACK_TYPE__(Packet) * to, const Packet& from, const Index n,
1730 const Index offset) {
1731 const Index packet_size = unpacket_traits<Packet>::size;
1732 eigen_internal_assert(n + offset <= packet_size && "number of elements plus offset will write past end of packet");
1733 const Index size = sizeof(__UNPACK_TYPE__(Packet));
1734#ifdef _ARCH_PWR9
1735 EIGEN_UNUSED_VARIABLE(packet_size);
1736 EIGEN_DEBUG_UNALIGNED_STORE
1737 Packet store = from;
1738 if (offset) {
1739 Packet16uc shift = pset1<Packet16uc>(offset * 8 * size);
1740#ifdef _BIG_ENDIAN
1741 store = Packet(vec_slo(Packet16uc(store), shift));
1742#else
1743 store = Packet(vec_sro(Packet16uc(store), shift));
1744#endif
1745 }
1746 vec_xst_len(store, to, n * size);
1747#else
1748 if (n) {
1749 EIGEN_ALIGN16 __UNPACK_TYPE__(Packet) store[packet_size];
1750 pstore(store, from);
1751 unsigned char* store2 = reinterpret_cast<unsigned char*>(store + offset);
1752 unsigned char* to2 = reinterpret_cast<unsigned char*>(to);
1753 Index n2 = n * size;
1754 if (16 <= n2) {
1755 pstoreu(to2, ploadu<Packet16uc>(store2));
1756 } else {
1757 memcpy((void*)to2, (void*)store2, n2);
1758 }
1759 }
1760#endif
1761}
1762
1763template <>
1764EIGEN_ALWAYS_INLINE void pstoreu_partial<float>(float* to, const Packet4f& from, const Index n, const Index offset) {
1765 pstoreu_partial_common<Packet4f>(to, from, n, offset);
1766}
1767template <>
1768EIGEN_ALWAYS_INLINE void pstoreu_partial<int>(int* to, const Packet4i& from, const Index n, const Index offset) {
1769 pstoreu_partial_common<Packet4i>(to, from, n, offset);
1770}
1771template <>
1772EIGEN_ALWAYS_INLINE void pstoreu_partial<short int>(short int* to, const Packet8s& from, const Index n,
1773 const Index offset) {
1774 pstoreu_partial_common<Packet8s>(to, from, n, offset);
1775}
1776template <>
1777EIGEN_ALWAYS_INLINE void pstoreu_partial<unsigned short int>(unsigned short int* to, const Packet8us& from,
1778 const Index n, const Index offset) {
1779 pstoreu_partial_common<Packet8us>(to, from, n, offset);
1780}
1781template <>
1782EIGEN_ALWAYS_INLINE void pstoreu_partial<bfloat16>(bfloat16* to, const Packet8bf& from, const Index n,
1783 const Index offset) {
1784 pstoreu_partial_common<Packet8us>(reinterpret_cast<unsigned short int*>(to), from, n, offset);
1785}
1786template <>
1787EIGEN_ALWAYS_INLINE void pstoreu_partial<signed char>(signed char* to, const Packet16c& from, const Index n,
1788 const Index offset) {
1789 pstoreu_partial_common<Packet16c>(to, from, n, offset);
1790}
1791template <>
1792EIGEN_ALWAYS_INLINE void pstoreu_partial<unsigned char>(unsigned char* to, const Packet16uc& from, const Index n,
1793 const Index offset) {
1794 pstoreu_partial_common<Packet16uc>(to, from, n, offset);
1795}
1796
1797template <>
1798EIGEN_STRONG_INLINE void prefetch<float>(const float* addr) {
1799 EIGEN_PPC_PREFETCH(addr);
1800}
1801template <>
1802EIGEN_STRONG_INLINE void prefetch<int>(const int* addr) {
1803 EIGEN_PPC_PREFETCH(addr);
1804}
1805
1806template <>
1807EIGEN_STRONG_INLINE float pfirst<Packet4f>(const Packet4f& a) {
1808 EIGEN_ALIGN16 float x;
1809 vec_ste(a, 0, &x);
1810 return x;
1811}
1812template <>
1813EIGEN_STRONG_INLINE int pfirst<Packet4i>(const Packet4i& a) {
1814 EIGEN_ALIGN16 int x;
1815 vec_ste(a, 0, &x);
1816 return x;
1817}
1818
1819template <typename Packet>
1820EIGEN_STRONG_INLINE __UNPACK_TYPE__(Packet) pfirst_common(const Packet& a) {
1821 EIGEN_ALIGN16 __UNPACK_TYPE__(Packet) x;
1822 vec_ste(a, 0, &x);
1823 return x;
1824}
1825
1826template <>
1827EIGEN_STRONG_INLINE short int pfirst<Packet8s>(const Packet8s& a) {
1828 return pfirst_common<Packet8s>(a);
1829}
1830
1831template <>
1832EIGEN_STRONG_INLINE unsigned short int pfirst<Packet8us>(const Packet8us& a) {
1833 return pfirst_common<Packet8us>(a);
1834}
1835
1836template <>
1837EIGEN_STRONG_INLINE signed char pfirst<Packet16c>(const Packet16c& a) {
1838 return pfirst_common<Packet16c>(a);
1839}
1840
1841template <>
1842EIGEN_STRONG_INLINE unsigned char pfirst<Packet16uc>(const Packet16uc& a) {
1843 return pfirst_common<Packet16uc>(a);
1844}
1845
1846template <>
1847EIGEN_STRONG_INLINE Packet4f preverse(const Packet4f& a) {
1848 return reinterpret_cast<Packet4f>(
1849 vec_perm(reinterpret_cast<Packet16uc>(a), reinterpret_cast<Packet16uc>(a), p16uc_REVERSE32));
1850}
1851template <>
1852EIGEN_STRONG_INLINE Packet4i preverse(const Packet4i& a) {
1853 return reinterpret_cast<Packet4i>(
1854 vec_perm(reinterpret_cast<Packet16uc>(a), reinterpret_cast<Packet16uc>(a), p16uc_REVERSE32));
1855}
1856template <>
1857EIGEN_STRONG_INLINE Packet8s preverse(const Packet8s& a) {
1858 return reinterpret_cast<Packet8s>(
1859 vec_perm(reinterpret_cast<Packet16uc>(a), reinterpret_cast<Packet16uc>(a), p16uc_REVERSE16));
1860}
1861template <>
1862EIGEN_STRONG_INLINE Packet8us preverse(const Packet8us& a) {
1863 return reinterpret_cast<Packet8us>(
1864 vec_perm(reinterpret_cast<Packet16uc>(a), reinterpret_cast<Packet16uc>(a), p16uc_REVERSE16));
1865}
1866template <>
1867EIGEN_STRONG_INLINE Packet16c preverse(const Packet16c& a) {
1868 return vec_perm(a, a, p16uc_REVERSE8);
1869}
1870template <>
1871EIGEN_STRONG_INLINE Packet16uc preverse(const Packet16uc& a) {
1872 return vec_perm(a, a, p16uc_REVERSE8);
1873}
1874template <>
1875EIGEN_STRONG_INLINE Packet8bf preverse(const Packet8bf& a) {
1876 return preverse<Packet8us>(a);
1877}
1878
1879template <>
1880EIGEN_STRONG_INLINE Packet4f pabs(const Packet4f& a) {
1881 return vec_abs(a);
1882}
1883template <>
1884EIGEN_STRONG_INLINE Packet4i pabs(const Packet4i& a) {
1885 return vec_abs(a);
1886}
1887template <>
1888EIGEN_STRONG_INLINE Packet8s pabs(const Packet8s& a) {
1889 return vec_abs(a);
1890}
1891template <>
1892EIGEN_STRONG_INLINE Packet16c pabs(const Packet16c& a) {
1893 return vec_abs(a);
1894}
1895template <>
1896EIGEN_STRONG_INLINE Packet8bf pabs(const Packet8bf& a) {
1897 EIGEN_DECLARE_CONST_FAST_Packet8us(abs_mask, 0x7FFF);
1898 return pand<Packet8us>(p8us_abs_mask, a);
1899}
1900
1901template <>
1902EIGEN_STRONG_INLINE Packet8bf psignbit(const Packet8bf& a) {
1903 return vec_sra(a.m_val, vec_splat_u16(15));
1904}
1905template <>
1906EIGEN_STRONG_INLINE Packet4f psignbit(const Packet4f& a) {
1907 return (Packet4f)vec_sra((Packet4i)a, vec_splats((unsigned int)(31)));
1908}
1909
1910template <int N>
1911EIGEN_STRONG_INLINE Packet4i parithmetic_shift_right(const Packet4i& a) {
1912 return vec_sra(a, reinterpret_cast<Packet4ui>(pset1<Packet4i>(N)));
1913}
1914template <int N>
1915EIGEN_STRONG_INLINE Packet4i plogical_shift_right(const Packet4i& a) {
1916 return vec_sr(a, reinterpret_cast<Packet4ui>(pset1<Packet4i>(N)));
1917}
1918template <int N>
1919EIGEN_STRONG_INLINE Packet4i plogical_shift_left(const Packet4i& a) {
1920 return vec_sl(a, reinterpret_cast<Packet4ui>(pset1<Packet4i>(N)));
1921}
1922template <int N>
1923EIGEN_STRONG_INLINE Packet16c parithmetic_shift_right(const Packet16c& a) {
1924 return vec_sra(a, pset1<Packet16uc>(static_cast<unsigned char>(N)));
1925}
1926template <int N>
1927EIGEN_STRONG_INLINE Packet16c plogical_shift_right(const Packet16c& a) {
1928 return vec_sr(a, pset1<Packet16uc>(static_cast<unsigned char>(N)));
1929}
1930template <int N>
1931EIGEN_STRONG_INLINE Packet16c plogical_shift_left(const Packet16c& a) {
1932 return vec_sl(a, pset1<Packet16uc>(static_cast<unsigned char>(N)));
1933}
1934template <int N>
1935EIGEN_STRONG_INLINE Packet16uc parithmetic_shift_right(const Packet16uc& a) {
1936 return vec_sr(a, pset1<Packet16uc>(static_cast<unsigned char>(N)));
1937}
1938template <int N>
1939EIGEN_STRONG_INLINE Packet16uc plogical_shift_right(const Packet16uc& a) {
1940 return vec_sr(a, pset1<Packet16uc>(static_cast<unsigned char>(N)));
1941}
1942template <int N>
1943EIGEN_STRONG_INLINE Packet16uc plogical_shift_left(const Packet16uc& a) {
1944 return vec_sl(a, pset1<Packet16uc>(static_cast<unsigned char>(N)));
1945}
1946template <int N>
1947EIGEN_STRONG_INLINE Packet8s parithmetic_shift_right(const Packet8s& a) {
1948 const EIGEN_DECLARE_CONST_FAST_Packet8us(mask, N);
1949 return vec_sra(a, p8us_mask);
1950}
1951template <int N>
1952EIGEN_STRONG_INLINE Packet8s plogical_shift_right(const Packet8s& a) {
1953 const EIGEN_DECLARE_CONST_FAST_Packet8us(mask, N);
1954 return vec_sr(a, p8us_mask);
1955}
1956template <int N>
1957EIGEN_STRONG_INLINE Packet8s plogical_shift_left(const Packet8s& a) {
1958 const EIGEN_DECLARE_CONST_FAST_Packet8us(mask, N);
1959 return vec_sl(a, p8us_mask);
1960}
1961template <int N>
1962EIGEN_STRONG_INLINE Packet8us parithmetic_shift_right(const Packet8us& a) {
1963 const EIGEN_DECLARE_CONST_FAST_Packet8us(mask, N);
1964 return vec_sr(a, p8us_mask);
1965}
1966template <int N>
1967EIGEN_STRONG_INLINE Packet8us plogical_shift_right(const Packet8us& a) {
1968 const EIGEN_DECLARE_CONST_FAST_Packet8us(mask, N);
1969 return vec_sr(a, p8us_mask);
1970}
1971template <int N>
1972EIGEN_STRONG_INLINE Packet8us plogical_shift_left(const Packet8us& a) {
1973 const EIGEN_DECLARE_CONST_FAST_Packet8us(mask, N);
1974 return vec_sl(a, p8us_mask);
1975}
1976template <int N>
1977EIGEN_STRONG_INLINE Packet4f plogical_shift_left(const Packet4f& a) {
1978 const EIGEN_DECLARE_CONST_FAST_Packet4ui(mask, N);
1979 Packet4ui r = vec_sl(reinterpret_cast<Packet4ui>(a), p4ui_mask);
1980 return reinterpret_cast<Packet4f>(r);
1981}
1982
1983template <int N>
1984EIGEN_STRONG_INLINE Packet4f plogical_shift_right(const Packet4f& a) {
1985 const EIGEN_DECLARE_CONST_FAST_Packet4ui(mask, N);
1986 Packet4ui r = vec_sr(reinterpret_cast<Packet4ui>(a), p4ui_mask);
1987 return reinterpret_cast<Packet4f>(r);
1988}
1989
1990template <int N>
1991EIGEN_STRONG_INLINE Packet4ui parithmetic_shift_right(const Packet4ui& a) {
1992 const EIGEN_DECLARE_CONST_FAST_Packet4ui(mask, N);
1993 return vec_sr(a, p4ui_mask);
1994}
1995
1996template <int N>
1997EIGEN_STRONG_INLINE Packet4ui plogical_shift_right(const Packet4ui& a) {
1998 const EIGEN_DECLARE_CONST_FAST_Packet4ui(mask, N);
1999 return vec_sr(a, p4ui_mask);
2000}
2001
2002template <int N>
2003EIGEN_STRONG_INLINE Packet4ui plogical_shift_left(const Packet4ui& a) {
2004 const EIGEN_DECLARE_CONST_FAST_Packet4ui(mask, N);
2005 return vec_sl(a, p4ui_mask);
2006}
2007
2008EIGEN_STRONG_INLINE Packet4f Bf16ToF32Even(const Packet8bf& bf) {
2009 return plogical_shift_left<16>(reinterpret_cast<Packet4f>(bf.m_val));
2010}
2011
2012EIGEN_STRONG_INLINE Packet4f Bf16ToF32Odd(const Packet8bf& bf) {
2013 const EIGEN_DECLARE_CONST_FAST_Packet4ui(high_mask, 0xFFFF0000);
2014 return pand<Packet4f>(reinterpret_cast<Packet4f>(bf.m_val), reinterpret_cast<Packet4f>(p4ui_high_mask));
2015}
2016
2017EIGEN_ALWAYS_INLINE Packet8us pmerge(const Packet4ui& even, const Packet4ui& odd) {
2018#ifdef _BIG_ENDIAN
2019 return vec_perm(reinterpret_cast<Packet8us>(odd), reinterpret_cast<Packet8us>(even), p16uc_MERGEO16);
2020#else
2021 return vec_perm(reinterpret_cast<Packet8us>(even), reinterpret_cast<Packet8us>(odd), p16uc_MERGEE16);
2022#endif
2023}
2024
2025// Simple interleaving of bool masks, prevents true values from being
2026// converted to NaNs.
2027EIGEN_STRONG_INLINE Packet8bf F32ToBf16Bool(const Packet4f& even, const Packet4f& odd) {
2028 return pmerge(reinterpret_cast<Packet4ui>(even), reinterpret_cast<Packet4ui>(odd));
2029}
2030
2031// #define SUPPORT_BF16_SUBNORMALS
2032
2033#ifndef __VEC_CLASS_FP_NAN
2034#define __VEC_CLASS_FP_NAN (1 << 6)
2035#endif
2036
2037#if defined(SUPPORT_BF16_SUBNORMALS) && !defined(__VEC_CLASS_FP_SUBNORMAL)
2038#define __VEC_CLASS_FP_SUBNORMAL_P (1 << 1)
2039#define __VEC_CLASS_FP_SUBNORMAL_N (1 << 0)
2040
2041#define __VEC_CLASS_FP_SUBNORMAL (__VEC_CLASS_FP_SUBNORMAL_P | __VEC_CLASS_FP_SUBNORMAL_N)
2042#endif
2043
2044EIGEN_STRONG_INLINE Packet8bf F32ToBf16(const Packet4f& p4f) {
2045#ifdef _ARCH_PWR10
2046 return reinterpret_cast<Packet8us>(__builtin_vsx_xvcvspbf16(reinterpret_cast<Packet16uc>(p4f)));
2047#else
2048 Packet4ui input = reinterpret_cast<Packet4ui>(p4f);
2049 Packet4ui lsb = plogical_shift_right<16>(input);
2050 lsb = pand<Packet4ui>(lsb, reinterpret_cast<Packet4ui>(p4i_ONE));
2051
2052 EIGEN_DECLARE_CONST_FAST_Packet4ui(BIAS, 0x7FFFu);
2053 Packet4ui rounding_bias = padd<Packet4ui>(lsb, p4ui_BIAS);
2054 input = padd<Packet4ui>(input, rounding_bias);
2055
2056 const EIGEN_DECLARE_CONST_FAST_Packet4ui(nan, 0x7FC00000);
2057#if defined(_ARCH_PWR9) && defined(EIGEN_VECTORIZE_VSX)
2058 Packet4bi nan_selector = vec_test_data_class(p4f, __VEC_CLASS_FP_NAN);
2059 input = vec_sel(input, p4ui_nan, nan_selector);
2060
2061#ifdef SUPPORT_BF16_SUBNORMALS
2062 Packet4bi subnormal_selector = vec_test_data_class(p4f, __VEC_CLASS_FP_SUBNORMAL);
2063 input = vec_sel(input, reinterpret_cast<Packet4ui>(p4f), subnormal_selector);
2064#endif
2065#else
2066#ifdef SUPPORT_BF16_SUBNORMALS
2067 // Test NaN and Subnormal
2068 const EIGEN_DECLARE_CONST_FAST_Packet4ui(exp_mask, 0x7F800000);
2069 Packet4ui exp = pand<Packet4ui>(p4ui_exp_mask, reinterpret_cast<Packet4ui>(p4f));
2070
2071 const EIGEN_DECLARE_CONST_FAST_Packet4ui(mantissa_mask, 0x7FFFFF);
2072 Packet4ui mantissa = pand<Packet4ui>(p4ui_mantissa_mask, reinterpret_cast<Packet4ui>(p4f));
2073
2074 Packet4bi is_max_exp = vec_cmpeq(exp, p4ui_exp_mask);
2075 Packet4bi is_mant_zero = vec_cmpeq(mantissa, reinterpret_cast<Packet4ui>(p4i_ZERO));
2076
2077 Packet4ui nan_selector =
2078 pandnot<Packet4ui>(reinterpret_cast<Packet4ui>(is_max_exp), reinterpret_cast<Packet4ui>(is_mant_zero));
2079
2080 Packet4bi is_zero_exp = vec_cmpeq(exp, reinterpret_cast<Packet4ui>(p4i_ZERO));
2081
2082 Packet4ui subnormal_selector =
2083 pandnot<Packet4ui>(reinterpret_cast<Packet4ui>(is_zero_exp), reinterpret_cast<Packet4ui>(is_mant_zero));
2084
2085 input = vec_sel(input, p4ui_nan, nan_selector);
2086 input = vec_sel(input, reinterpret_cast<Packet4ui>(p4f), subnormal_selector);
2087#else
2088 // Test only NaN
2089 Packet4bi nan_selector = vec_cmpeq(p4f, p4f);
2090
2091 input = vec_sel(p4ui_nan, input, nan_selector);
2092#endif
2093#endif
2094
2095 input = plogical_shift_right<16>(input);
2096 return reinterpret_cast<Packet8us>(input);
2097#endif
2098}
2099
2100#ifdef _BIG_ENDIAN
2106template <bool lohi>
2107EIGEN_ALWAYS_INLINE Packet8bf Bf16PackHigh(const Packet4f& lo, const Packet4f& hi) {
2108 if (lohi) {
2109 return vec_perm(reinterpret_cast<Packet8us>(lo), reinterpret_cast<Packet8us>(hi), p16uc_MERGEH16);
2110 } else {
2111 return vec_perm(reinterpret_cast<Packet8us>(hi), reinterpret_cast<Packet8us>(lo), p16uc_MERGEE16);
2112 }
2113}
2114
2120template <bool lohi>
2121EIGEN_ALWAYS_INLINE Packet8bf Bf16PackLow(const Packet4f& lo, const Packet4f& hi) {
2122 if (lohi) {
2123 return vec_pack(reinterpret_cast<Packet4ui>(lo), reinterpret_cast<Packet4ui>(hi));
2124 } else {
2125 return vec_perm(reinterpret_cast<Packet8us>(hi), reinterpret_cast<Packet8us>(lo), p16uc_MERGEO16);
2126 }
2127}
2128#else
2129template <bool lohi>
2130EIGEN_ALWAYS_INLINE Packet8bf Bf16PackLow(const Packet4f& hi, const Packet4f& lo) {
2131 if (lohi) {
2132 return vec_pack(reinterpret_cast<Packet4ui>(hi), reinterpret_cast<Packet4ui>(lo));
2133 } else {
2134 return vec_perm(reinterpret_cast<Packet8us>(hi), reinterpret_cast<Packet8us>(lo), p16uc_MERGEE16);
2135 }
2136}
2137
2138template <bool lohi>
2139EIGEN_ALWAYS_INLINE Packet8bf Bf16PackHigh(const Packet4f& hi, const Packet4f& lo) {
2140 if (lohi) {
2141 return vec_perm(reinterpret_cast<Packet8us>(hi), reinterpret_cast<Packet8us>(lo), p16uc_MERGEL16);
2142 } else {
2143 return vec_perm(reinterpret_cast<Packet8us>(hi), reinterpret_cast<Packet8us>(lo), p16uc_MERGEO16);
2144 }
2145}
2146#endif
2147
2153template <bool lohi = true>
2154EIGEN_ALWAYS_INLINE Packet8bf F32ToBf16Two(const Packet4f& lo, const Packet4f& hi) {
2155 Packet8us p4f = Bf16PackHigh<lohi>(lo, hi);
2156 Packet8us p4f2 = Bf16PackLow<lohi>(lo, hi);
2157
2158 Packet8us lsb = pand<Packet8us>(p4f, p8us_ONE);
2159 EIGEN_DECLARE_CONST_FAST_Packet8us(BIAS, 0x7FFFu);
2160 lsb = padd<Packet8us>(lsb, p8us_BIAS);
2161 lsb = padd<Packet8us>(lsb, p4f2);
2162
2163 Packet8bi rounding_bias = vec_cmplt(lsb, p4f2);
2164 Packet8us input = psub<Packet8us>(p4f, reinterpret_cast<Packet8us>(rounding_bias));
2165
2166#if defined(_ARCH_PWR9) && defined(EIGEN_VECTORIZE_VSX)
2167 Packet4bi nan_selector_lo = vec_test_data_class(lo, __VEC_CLASS_FP_NAN);
2168 Packet4bi nan_selector_hi = vec_test_data_class(hi, __VEC_CLASS_FP_NAN);
2169 Packet8us nan_selector =
2170 Bf16PackLow<lohi>(reinterpret_cast<Packet4f>(nan_selector_lo), reinterpret_cast<Packet4f>(nan_selector_hi));
2171
2172 input = vec_sel(input, p8us_BIAS, nan_selector);
2173
2174#ifdef SUPPORT_BF16_SUBNORMALS
2175 Packet4bi subnormal_selector_lo = vec_test_data_class(lo, __VEC_CLASS_FP_SUBNORMAL);
2176 Packet4bi subnormal_selector_hi = vec_test_data_class(hi, __VEC_CLASS_FP_SUBNORMAL);
2177 Packet8us subnormal_selector = Bf16PackLow<lohi>(reinterpret_cast<Packet4f>(subnormal_selector_lo),
2178 reinterpret_cast<Packet4f>(subnormal_selector_hi));
2179
2180 input = vec_sel(input, reinterpret_cast<Packet8us>(p4f), subnormal_selector);
2181#endif
2182#else
2183#ifdef SUPPORT_BF16_SUBNORMALS
2184 // Test NaN and Subnormal
2185 const EIGEN_DECLARE_CONST_FAST_Packet8us(exp_mask, 0x7F80);
2186 Packet8us exp = pand<Packet8us>(p8us_exp_mask, p4f);
2187
2188 const EIGEN_DECLARE_CONST_FAST_Packet8us(mantissa_mask, 0x7Fu);
2189 Packet8us mantissa = pand<Packet8us>(p8us_mantissa_mask, p4f);
2190
2191 Packet8bi is_max_exp = vec_cmpeq(exp, p8us_exp_mask);
2192 Packet8bi is_mant_zero = vec_cmpeq(mantissa, reinterpret_cast<Packet8us>(p4i_ZERO));
2193
2194 Packet8us nan_selector =
2195 pandnot<Packet8us>(reinterpret_cast<Packet8us>(is_max_exp), reinterpret_cast<Packet8us>(is_mant_zero));
2196
2197 Packet8bi is_zero_exp = vec_cmpeq(exp, reinterpret_cast<Packet8us>(p4i_ZERO));
2198
2199 Packet8us subnormal_selector =
2200 pandnot<Packet8us>(reinterpret_cast<Packet8us>(is_zero_exp), reinterpret_cast<Packet8us>(is_mant_zero));
2201
2202 // Using BIAS as NaN (since any or all of the last 7 bits can be set)
2203 input = vec_sel(input, p8us_BIAS, nan_selector);
2204 input = vec_sel(input, reinterpret_cast<Packet8us>(p4f), subnormal_selector);
2205#else
2206 // Test only NaN
2207 Packet4bi nan_selector_lo = vec_cmpeq(lo, lo);
2208 Packet4bi nan_selector_hi = vec_cmpeq(hi, hi);
2209 Packet8us nan_selector =
2210 Bf16PackLow<lohi>(reinterpret_cast<Packet4f>(nan_selector_lo), reinterpret_cast<Packet4f>(nan_selector_hi));
2211
2212 input = vec_sel(p8us_BIAS, input, nan_selector);
2213#endif
2214#endif
2215
2216 return input;
2217}
2218
2222EIGEN_STRONG_INLINE Packet8bf F32ToBf16Both(const Packet4f& lo, const Packet4f& hi) {
2223#ifdef _ARCH_PWR10
2224 Packet8bf fp16_0 = F32ToBf16(lo);
2225 Packet8bf fp16_1 = F32ToBf16(hi);
2226 return vec_pack(reinterpret_cast<Packet4ui>(fp16_0.m_val), reinterpret_cast<Packet4ui>(fp16_1.m_val));
2227#else
2228 return F32ToBf16Two(lo, hi);
2229#endif
2230}
2231
2235EIGEN_STRONG_INLINE Packet8bf F32ToBf16(const Packet4f& even, const Packet4f& odd) {
2236#ifdef _ARCH_PWR10
2237 return pmerge(reinterpret_cast<Packet4ui>(F32ToBf16(even).m_val), reinterpret_cast<Packet4ui>(F32ToBf16(odd).m_val));
2238#else
2239 return F32ToBf16Two<false>(even, odd);
2240#endif
2241}
2242#define BF16_TO_F32_UNARY_OP_WRAPPER(OP, A) \
2243 Packet4f a_even = Bf16ToF32Even(A); \
2244 Packet4f a_odd = Bf16ToF32Odd(A); \
2245 Packet4f op_even = OP(a_even); \
2246 Packet4f op_odd = OP(a_odd); \
2247 return F32ToBf16(op_even, op_odd);
2248
2249#define BF16_TO_F32_BINARY_OP_WRAPPER(OP, A, B) \
2250 Packet4f a_even = Bf16ToF32Even(A); \
2251 Packet4f a_odd = Bf16ToF32Odd(A); \
2252 Packet4f b_even = Bf16ToF32Even(B); \
2253 Packet4f b_odd = Bf16ToF32Odd(B); \
2254 Packet4f op_even = OP(a_even, b_even); \
2255 Packet4f op_odd = OP(a_odd, b_odd); \
2256 return F32ToBf16(op_even, op_odd);
2257
2258#define BF16_TO_F32_BINARY_OP_WRAPPER_BOOL(OP, A, B) \
2259 Packet4f a_even = Bf16ToF32Even(A); \
2260 Packet4f a_odd = Bf16ToF32Odd(A); \
2261 Packet4f b_even = Bf16ToF32Even(B); \
2262 Packet4f b_odd = Bf16ToF32Odd(B); \
2263 Packet4f op_even = OP(a_even, b_even); \
2264 Packet4f op_odd = OP(a_odd, b_odd); \
2265 return F32ToBf16Bool(op_even, op_odd);
2266
2267template <>
2268EIGEN_STRONG_INLINE Packet8bf padd<Packet8bf>(const Packet8bf& a, const Packet8bf& b) {
2269 BF16_TO_F32_BINARY_OP_WRAPPER(padd<Packet4f>, a, b);
2270}
2271
2272template <>
2273EIGEN_STRONG_INLINE Packet8bf pmul<Packet8bf>(const Packet8bf& a, const Packet8bf& b) {
2274 BF16_TO_F32_BINARY_OP_WRAPPER(pmul<Packet4f>, a, b);
2275}
2276
2277template <>
2278EIGEN_STRONG_INLINE Packet8bf pdiv<Packet8bf>(const Packet8bf& a, const Packet8bf& b) {
2279 BF16_TO_F32_BINARY_OP_WRAPPER(pdiv<Packet4f>, a, b);
2280}
2281
2282template <>
2283EIGEN_STRONG_INLINE Packet8bf pnegate<Packet8bf>(const Packet8bf& a) {
2284 EIGEN_DECLARE_CONST_FAST_Packet8us(neg_mask, 0x8000);
2285 return pxor<Packet8us>(p8us_neg_mask, a);
2286}
2287
2288template <>
2289EIGEN_STRONG_INLINE Packet8bf psub<Packet8bf>(const Packet8bf& a, const Packet8bf& b) {
2290 BF16_TO_F32_BINARY_OP_WRAPPER(psub<Packet4f>, a, b);
2291}
2292
2293template <>
2294EIGEN_STRONG_INLINE Packet8bf pexp<Packet8bf>(const Packet8bf& a) {
2295 BF16_TO_F32_UNARY_OP_WRAPPER(pexp_float, a);
2296}
2297
2298template <>
2299EIGEN_STRONG_INLINE Packet8bf pexp2<Packet8bf>(const Packet8bf& a) {
2300 BF16_TO_F32_UNARY_OP_WRAPPER(generic_exp2, a);
2301}
2302
2303template <>
2304EIGEN_STRONG_INLINE Packet4f pldexp<Packet4f>(const Packet4f& a, const Packet4f& exponent) {
2305 return pldexp_generic(a, exponent);
2306}
2307template <>
2308EIGEN_STRONG_INLINE Packet8bf pldexp<Packet8bf>(const Packet8bf& a, const Packet8bf& exponent) {
2309 BF16_TO_F32_BINARY_OP_WRAPPER(pldexp<Packet4f>, a, exponent);
2310}
2311
2312template <>
2313EIGEN_STRONG_INLINE Packet4f pfrexp<Packet4f>(const Packet4f& a, Packet4f& exponent) {
2314 return pfrexp_generic(a, exponent);
2315}
2316template <>
2317EIGEN_STRONG_INLINE Packet8bf pfrexp<Packet8bf>(const Packet8bf& a, Packet8bf& e) {
2318 Packet4f a_even = Bf16ToF32Even(a);
2319 Packet4f a_odd = Bf16ToF32Odd(a);
2320 Packet4f e_even;
2321 Packet4f e_odd;
2322 Packet4f op_even = pfrexp<Packet4f>(a_even, e_even);
2323 Packet4f op_odd = pfrexp<Packet4f>(a_odd, e_odd);
2324 e = F32ToBf16(e_even, e_odd);
2325 return F32ToBf16(op_even, op_odd);
2326}
2327
2328template <>
2329EIGEN_STRONG_INLINE Packet8bf psin<Packet8bf>(const Packet8bf& a) {
2330 BF16_TO_F32_UNARY_OP_WRAPPER(psin_float, a);
2331}
2332template <>
2333EIGEN_STRONG_INLINE Packet8bf pcos<Packet8bf>(const Packet8bf& a) {
2334 BF16_TO_F32_UNARY_OP_WRAPPER(pcos_float, a);
2335}
2336template <>
2337EIGEN_STRONG_INLINE Packet8bf plog<Packet8bf>(const Packet8bf& a) {
2338 BF16_TO_F32_UNARY_OP_WRAPPER(plog_float, a);
2339}
2340template <>
2341EIGEN_STRONG_INLINE Packet8bf pfloor<Packet8bf>(const Packet8bf& a) {
2342 BF16_TO_F32_UNARY_OP_WRAPPER(pfloor<Packet4f>, a);
2343}
2344template <>
2345EIGEN_STRONG_INLINE Packet8bf pceil<Packet8bf>(const Packet8bf& a) {
2346 BF16_TO_F32_UNARY_OP_WRAPPER(pceil<Packet4f>, a);
2347}
2348template <>
2349EIGEN_STRONG_INLINE Packet8bf pround<Packet8bf>(const Packet8bf& a) {
2350 BF16_TO_F32_UNARY_OP_WRAPPER(pround<Packet4f>, a);
2351}
2352template <>
2353EIGEN_STRONG_INLINE Packet8bf ptrunc<Packet8bf>(const Packet8bf& a) {
2354 BF16_TO_F32_UNARY_OP_WRAPPER(ptrunc<Packet4f>, a);
2355}
2356#ifdef EIGEN_VECTORIZE_VSX
2357template <>
2358EIGEN_STRONG_INLINE Packet8bf print<Packet8bf>(const Packet8bf& a) {
2359 BF16_TO_F32_UNARY_OP_WRAPPER(print<Packet4f>, a);
2360}
2361#endif
2362template <>
2363EIGEN_STRONG_INLINE Packet8bf pmadd(const Packet8bf& a, const Packet8bf& b, const Packet8bf& c) {
2364 Packet4f a_even = Bf16ToF32Even(a);
2365 Packet4f a_odd = Bf16ToF32Odd(a);
2366 Packet4f b_even = Bf16ToF32Even(b);
2367 Packet4f b_odd = Bf16ToF32Odd(b);
2368 Packet4f c_even = Bf16ToF32Even(c);
2369 Packet4f c_odd = Bf16ToF32Odd(c);
2370 Packet4f pmadd_even = pmadd<Packet4f>(a_even, b_even, c_even);
2371 Packet4f pmadd_odd = pmadd<Packet4f>(a_odd, b_odd, c_odd);
2372 return F32ToBf16(pmadd_even, pmadd_odd);
2373}
2374
2375template <>
2376EIGEN_STRONG_INLINE Packet8bf pmsub(const Packet8bf& a, const Packet8bf& b, const Packet8bf& c) {
2377 Packet4f a_even = Bf16ToF32Even(a);
2378 Packet4f a_odd = Bf16ToF32Odd(a);
2379 Packet4f b_even = Bf16ToF32Even(b);
2380 Packet4f b_odd = Bf16ToF32Odd(b);
2381 Packet4f c_even = Bf16ToF32Even(c);
2382 Packet4f c_odd = Bf16ToF32Odd(c);
2383 Packet4f pmadd_even = pmsub<Packet4f>(a_even, b_even, c_even);
2384 Packet4f pmadd_odd = pmsub<Packet4f>(a_odd, b_odd, c_odd);
2385 return F32ToBf16(pmadd_even, pmadd_odd);
2386}
2387template <>
2388EIGEN_STRONG_INLINE Packet8bf pnmadd(const Packet8bf& a, const Packet8bf& b, const Packet8bf& c) {
2389 Packet4f a_even = Bf16ToF32Even(a);
2390 Packet4f a_odd = Bf16ToF32Odd(a);
2391 Packet4f b_even = Bf16ToF32Even(b);
2392 Packet4f b_odd = Bf16ToF32Odd(b);
2393 Packet4f c_even = Bf16ToF32Even(c);
2394 Packet4f c_odd = Bf16ToF32Odd(c);
2395 Packet4f pmadd_even = pnmadd<Packet4f>(a_even, b_even, c_even);
2396 Packet4f pmadd_odd = pnmadd<Packet4f>(a_odd, b_odd, c_odd);
2397 return F32ToBf16(pmadd_even, pmadd_odd);
2398}
2399
2400template <>
2401EIGEN_STRONG_INLINE Packet8bf pnmsub(const Packet8bf& a, const Packet8bf& b, const Packet8bf& c) {
2402 Packet4f a_even = Bf16ToF32Even(a);
2403 Packet4f a_odd = Bf16ToF32Odd(a);
2404 Packet4f b_even = Bf16ToF32Even(b);
2405 Packet4f b_odd = Bf16ToF32Odd(b);
2406 Packet4f c_even = Bf16ToF32Even(c);
2407 Packet4f c_odd = Bf16ToF32Odd(c);
2408 Packet4f pmadd_even = pnmsub<Packet4f>(a_even, b_even, c_even);
2409 Packet4f pmadd_odd = pnmsub<Packet4f>(a_odd, b_odd, c_odd);
2410 return F32ToBf16(pmadd_even, pmadd_odd);
2411}
2412
2413template <>
2414EIGEN_STRONG_INLINE Packet8bf pmin<Packet8bf>(const Packet8bf& a, const Packet8bf& b) {
2415 BF16_TO_F32_BINARY_OP_WRAPPER(pmin<Packet4f>, a, b);
2416}
2417
2418template <>
2419EIGEN_STRONG_INLINE Packet8bf pmax<Packet8bf>(const Packet8bf& a, const Packet8bf& b) {
2420 BF16_TO_F32_BINARY_OP_WRAPPER(pmax<Packet4f>, a, b);
2421}
2422
2423template <>
2424EIGEN_STRONG_INLINE Packet8bf pcmp_lt(const Packet8bf& a, const Packet8bf& b) {
2425 BF16_TO_F32_BINARY_OP_WRAPPER_BOOL(pcmp_lt<Packet4f>, a, b);
2426}
2427template <>
2428EIGEN_STRONG_INLINE Packet8bf pcmp_lt_or_nan(const Packet8bf& a, const Packet8bf& b) {
2429 BF16_TO_F32_BINARY_OP_WRAPPER_BOOL(pcmp_lt_or_nan<Packet4f>, a, b);
2430}
2431template <>
2432EIGEN_STRONG_INLINE Packet8bf pcmp_le(const Packet8bf& a, const Packet8bf& b) {
2433 BF16_TO_F32_BINARY_OP_WRAPPER_BOOL(pcmp_le<Packet4f>, a, b);
2434}
2435template <>
2436EIGEN_STRONG_INLINE Packet8bf pcmp_eq(const Packet8bf& a, const Packet8bf& b) {
2437 BF16_TO_F32_BINARY_OP_WRAPPER_BOOL(pcmp_eq<Packet4f>, a, b);
2438}
2439
2440// Compare encoded lanes: widening bf16 subnormals to float loses them under DAZ/FZ.
2441template <>
2442EIGEN_STRONG_INLINE Packet8bf psign<Packet8bf>(const Packet8bf& a) {
2443 const Packet8us magnitude = vec_and(a.m_val, vec_splats(static_cast<unsigned short int>(0x7fff)));
2444 const Packet8bi is_zero = vec_cmpeq(magnitude, vec_splats(static_cast<unsigned short int>(0)));
2445 const Packet8us keep =
2446 vec_or(reinterpret_cast<Packet8us>(vec_cmpgt(magnitude, vec_splats(static_cast<unsigned short int>(0x7f80)))),
2447 vec_splats(static_cast<unsigned short int>(0x8000)));
2448 const Packet8us value = vec_sel(vec_splats(static_cast<unsigned short int>(0x3f80)), a.m_val, keep);
2449 return Packet8bf(vec_andc(value, reinterpret_cast<Packet8us>(is_zero)));
2450}
2451
2452template <>
2453EIGEN_STRONG_INLINE bfloat16 pfirst(const Packet8bf& a) {
2454 return Eigen::bfloat16_impl::raw_uint16_to_bfloat16((pfirst<Packet8us>(a)));
2455}
2456
2457template <>
2458EIGEN_STRONG_INLINE Packet8bf ploaddup<Packet8bf>(const bfloat16* from) {
2459 return ploaddup<Packet8us>(reinterpret_cast<const unsigned short int*>(from));
2460}
2461
2462template <>
2463EIGEN_STRONG_INLINE Packet8bf plset<Packet8bf>(const bfloat16& a) {
2464 bfloat16 countdown[8] = {bfloat16(0), bfloat16(1), bfloat16(2), bfloat16(3),
2465 bfloat16(4), bfloat16(5), bfloat16(6), bfloat16(7)};
2466 return padd<Packet8bf>(pset1<Packet8bf>(a), pload<Packet8bf>(countdown));
2467}
2468
2469template <>
2470EIGEN_STRONG_INLINE float predux<Packet4f>(const Packet4f& a) {
2471 Packet4f b, sum;
2472 b = vec_sld(a, a, 8);
2473 sum = a + b;
2474 b = vec_sld(sum, sum, 4);
2475 sum += b;
2476 return pfirst(sum);
2477}
2478
2479template <>
2480EIGEN_STRONG_INLINE int predux<Packet4i>(const Packet4i& a) {
2481 Packet4i b, sum;
2482 b = vec_sld(a, a, 8);
2483 sum = a + b;
2484 b = vec_sld(sum, sum, 4);
2485 sum += b;
2486 return pfirst(sum);
2487}
2488
2489template <>
2490EIGEN_STRONG_INLINE bfloat16 predux<Packet8bf>(const Packet8bf& a) {
2491 float redux_even = predux<Packet4f>(Bf16ToF32Even(a));
2492 float redux_odd = predux<Packet4f>(Bf16ToF32Odd(a));
2493 float f32_result = redux_even + redux_odd;
2494 return bfloat16(f32_result);
2495}
2496template <typename Packet>
2497EIGEN_STRONG_INLINE __UNPACK_TYPE__(Packet) predux_size8(const Packet& a) {
2498 union {
2499 Packet v;
2500 __UNPACK_TYPE__(Packet) n[8];
2501 } vt;
2502 vt.v = a;
2503
2504 EIGEN_ALIGN16 int first_loader[4] = {vt.n[0], vt.n[1], vt.n[2], vt.n[3]};
2505 EIGEN_ALIGN16 int second_loader[4] = {vt.n[4], vt.n[5], vt.n[6], vt.n[7]};
2506 Packet4i first_half = pload<Packet4i>(first_loader);
2507 Packet4i second_half = pload<Packet4i>(second_loader);
2508
2509 return static_cast<__UNPACK_TYPE__(Packet)>(predux(first_half) + predux(second_half));
2510}
2511
2512template <>
2513EIGEN_STRONG_INLINE short int predux<Packet8s>(const Packet8s& a) {
2514 return predux_size8<Packet8s>(a);
2515}
2516
2517template <>
2518EIGEN_STRONG_INLINE unsigned short int predux<Packet8us>(const Packet8us& a) {
2519 return predux_size8<Packet8us>(a);
2520}
2521
2522template <typename Packet>
2523EIGEN_STRONG_INLINE __UNPACK_TYPE__(Packet) predux_size16(const Packet& a) {
2524 union {
2525 Packet v;
2526 __UNPACK_TYPE__(Packet) n[16];
2527 } vt;
2528 vt.v = a;
2529
2530 EIGEN_ALIGN16 int first_loader[4] = {vt.n[0], vt.n[1], vt.n[2], vt.n[3]};
2531 EIGEN_ALIGN16 int second_loader[4] = {vt.n[4], vt.n[5], vt.n[6], vt.n[7]};
2532 EIGEN_ALIGN16 int third_loader[4] = {vt.n[8], vt.n[9], vt.n[10], vt.n[11]};
2533 EIGEN_ALIGN16 int fourth_loader[4] = {vt.n[12], vt.n[13], vt.n[14], vt.n[15]};
2534
2535 Packet4i first_quarter = pload<Packet4i>(first_loader);
2536 Packet4i second_quarter = pload<Packet4i>(second_loader);
2537 Packet4i third_quarter = pload<Packet4i>(third_loader);
2538 Packet4i fourth_quarter = pload<Packet4i>(fourth_loader);
2539
2540 return static_cast<__UNPACK_TYPE__(Packet)>(predux(first_quarter) + predux(second_quarter) + predux(third_quarter) +
2541 predux(fourth_quarter));
2542}
2543
2544template <>
2545EIGEN_STRONG_INLINE signed char predux<Packet16c>(const Packet16c& a) {
2546 return predux_size16<Packet16c>(a);
2547}
2548
2549template <>
2550EIGEN_STRONG_INLINE unsigned char predux<Packet16uc>(const Packet16uc& a) {
2551 return predux_size16<Packet16uc>(a);
2552}
2553
2554// Other reduction functions:
2555// mul
2556template <>
2557EIGEN_STRONG_INLINE float predux_mul<Packet4f>(const Packet4f& a) {
2558 Packet4f prod;
2559 prod = pmul(a, vec_sld(a, a, 8));
2560 return pfirst(pmul(prod, vec_sld(prod, prod, 4)));
2561}
2562
2563template <>
2564EIGEN_STRONG_INLINE int predux_mul<Packet4i>(const Packet4i& a) {
2565 EIGEN_ALIGN16 int aux[4];
2566 pstore(aux, a);
2567 return aux[0] * aux[1] * aux[2] * aux[3];
2568}
2569
2570template <>
2571EIGEN_STRONG_INLINE short int predux_mul<Packet8s>(const Packet8s& a) {
2572 Packet8s pair, quad, octo;
2573
2574 pair = vec_mul(a, vec_sld(a, a, 8));
2575 quad = vec_mul(pair, vec_sld(pair, pair, 4));
2576 octo = vec_mul(quad, vec_sld(quad, quad, 2));
2577
2578 return pfirst(octo);
2579}
2580
2581template <>
2582EIGEN_STRONG_INLINE unsigned short int predux_mul<Packet8us>(const Packet8us& a) {
2583 Packet8us pair, quad, octo;
2584
2585 pair = vec_mul(a, vec_sld(a, a, 8));
2586 quad = vec_mul(pair, vec_sld(pair, pair, 4));
2587 octo = vec_mul(quad, vec_sld(quad, quad, 2));
2588
2589 return pfirst(octo);
2590}
2591
2592template <>
2593EIGEN_STRONG_INLINE bfloat16 predux_mul<Packet8bf>(const Packet8bf& a) {
2594 float redux_even = predux_mul<Packet4f>(Bf16ToF32Even(a));
2595 float redux_odd = predux_mul<Packet4f>(Bf16ToF32Odd(a));
2596 float f32_result = redux_even * redux_odd;
2597 return bfloat16(f32_result);
2598}
2599
2600template <>
2601EIGEN_STRONG_INLINE signed char predux_mul<Packet16c>(const Packet16c& a) {
2602 Packet16c pair, quad, octo, result;
2603
2604 pair = vec_mul(a, vec_sld(a, a, 8));
2605 quad = vec_mul(pair, vec_sld(pair, pair, 4));
2606 octo = vec_mul(quad, vec_sld(quad, quad, 2));
2607 result = vec_mul(octo, vec_sld(octo, octo, 1));
2608
2609 return pfirst(result);
2610}
2611
2612template <>
2613EIGEN_STRONG_INLINE unsigned char predux_mul<Packet16uc>(const Packet16uc& a) {
2614 Packet16uc pair, quad, octo, result;
2615
2616 pair = vec_mul(a, vec_sld(a, a, 8));
2617 quad = vec_mul(pair, vec_sld(pair, pair, 4));
2618 octo = vec_mul(quad, vec_sld(quad, quad, 2));
2619 result = vec_mul(octo, vec_sld(octo, octo, 1));
2620
2621 return pfirst(result);
2622}
2623
2624// min
2625template <typename Packet>
2626EIGEN_STRONG_INLINE __UNPACK_TYPE__(Packet) predux_min4(const Packet& a) {
2627 Packet b, res;
2628 b = vec_min(a, vec_sld(a, a, 8));
2629 res = vec_min(b, vec_sld(b, b, 4));
2630 return pfirst(res);
2631}
2632
2633template <>
2634EIGEN_STRONG_INLINE float predux_min<Packet4f>(const Packet4f& a) {
2635 return predux_min4<Packet4f>(a);
2636}
2637
2638template <>
2639EIGEN_STRONG_INLINE int predux_min<Packet4i>(const Packet4i& a) {
2640 return predux_min4<Packet4i>(a);
2641}
2642
2643template <>
2644EIGEN_STRONG_INLINE bfloat16 predux_min<Packet8bf>(const Packet8bf& a) {
2645 float redux_even = predux_min<Packet4f>(Bf16ToF32Even(a));
2646 float redux_odd = predux_min<Packet4f>(Bf16ToF32Odd(a));
2647 float f32_result = (std::min)(redux_even, redux_odd);
2648 return bfloat16(f32_result);
2649}
2650
2651template <>
2652EIGEN_STRONG_INLINE short int predux_min<Packet8s>(const Packet8s& a) {
2653 Packet8s pair, quad, octo;
2654
2655 // pair = { Min(a0,a4), Min(a1,a5), Min(a2,a6), Min(a3,a7) }
2656 pair = vec_min(a, vec_sld(a, a, 8));
2657
2658 // quad = { Min(a0, a4, a2, a6), Min(a1, a5, a3, a7) }
2659 quad = vec_min(pair, vec_sld(pair, pair, 4));
2660
2661 // octo = { Min(a0, a4, a2, a6, a1, a5, a3, a7) }
2662 octo = vec_min(quad, vec_sld(quad, quad, 2));
2663 return pfirst(octo);
2664}
2665
2666template <>
2667EIGEN_STRONG_INLINE unsigned short int predux_min<Packet8us>(const Packet8us& a) {
2668 Packet8us pair, quad, octo;
2669
2670 // pair = { Min(a0,a4), Min(a1,a5), Min(a2,a6), Min(a3,a7) }
2671 pair = vec_min(a, vec_sld(a, a, 8));
2672
2673 // quad = { Min(a0, a4, a2, a6), Min(a1, a5, a3, a7) }
2674 quad = vec_min(pair, vec_sld(pair, pair, 4));
2675
2676 // octo = { Min(a0, a4, a2, a6, a1, a5, a3, a7) }
2677 octo = vec_min(quad, vec_sld(quad, quad, 2));
2678 return pfirst(octo);
2679}
2680
2681template <>
2682EIGEN_STRONG_INLINE signed char predux_min<Packet16c>(const Packet16c& a) {
2683 Packet16c pair, quad, octo, result;
2684
2685 pair = vec_min(a, vec_sld(a, a, 8));
2686 quad = vec_min(pair, vec_sld(pair, pair, 4));
2687 octo = vec_min(quad, vec_sld(quad, quad, 2));
2688 result = vec_min(octo, vec_sld(octo, octo, 1));
2689
2690 return pfirst(result);
2691}
2692
2693template <>
2694EIGEN_STRONG_INLINE unsigned char predux_min<Packet16uc>(const Packet16uc& a) {
2695 Packet16uc pair, quad, octo, result;
2696
2697 pair = vec_min(a, vec_sld(a, a, 8));
2698 quad = vec_min(pair, vec_sld(pair, pair, 4));
2699 octo = vec_min(quad, vec_sld(quad, quad, 2));
2700 result = vec_min(octo, vec_sld(octo, octo, 1));
2701
2702 return pfirst(result);
2703}
2704// max
2705template <typename Packet>
2706EIGEN_STRONG_INLINE __UNPACK_TYPE__(Packet) predux_max4(const Packet& a) {
2707 Packet b, res;
2708 b = vec_max(a, vec_sld(a, a, 8));
2709 res = vec_max(b, vec_sld(b, b, 4));
2710 return pfirst(res);
2711}
2712
2713template <>
2714EIGEN_STRONG_INLINE float predux_max<Packet4f>(const Packet4f& a) {
2715 return predux_max4<Packet4f>(a);
2716}
2717
2718template <>
2719EIGEN_STRONG_INLINE int predux_max<Packet4i>(const Packet4i& a) {
2720 return predux_max4<Packet4i>(a);
2721}
2722
2723template <>
2724EIGEN_STRONG_INLINE bfloat16 predux_max<Packet8bf>(const Packet8bf& a) {
2725 float redux_even = predux_max<Packet4f>(Bf16ToF32Even(a));
2726 float redux_odd = predux_max<Packet4f>(Bf16ToF32Odd(a));
2727 float f32_result = (std::max)(redux_even, redux_odd);
2728 return bfloat16(f32_result);
2729}
2730
2731template <>
2732EIGEN_STRONG_INLINE short int predux_max<Packet8s>(const Packet8s& a) {
2733 Packet8s pair, quad, octo;
2734
2735 // pair = { Max(a0,a4), Max(a1,a5), Max(a2,a6), Max(a3,a7) }
2736 pair = vec_max(a, vec_sld(a, a, 8));
2737
2738 // quad = { Max(a0, a4, a2, a6), Max(a1, a5, a3, a7) }
2739 quad = vec_max(pair, vec_sld(pair, pair, 4));
2740
2741 // octo = { Max(a0, a4, a2, a6, a1, a5, a3, a7) }
2742 octo = vec_max(quad, vec_sld(quad, quad, 2));
2743 return pfirst(octo);
2744}
2745
2746template <>
2747EIGEN_STRONG_INLINE unsigned short int predux_max<Packet8us>(const Packet8us& a) {
2748 Packet8us pair, quad, octo;
2749
2750 // pair = { Max(a0,a4), Max(a1,a5), Max(a2,a6), Max(a3,a7) }
2751 pair = vec_max(a, vec_sld(a, a, 8));
2752
2753 // quad = { Max(a0, a4, a2, a6), Max(a1, a5, a3, a7) }
2754 quad = vec_max(pair, vec_sld(pair, pair, 4));
2755
2756 // octo = { Max(a0, a4, a2, a6, a1, a5, a3, a7) }
2757 octo = vec_max(quad, vec_sld(quad, quad, 2));
2758 return pfirst(octo);
2759}
2760
2761template <>
2762EIGEN_STRONG_INLINE signed char predux_max<Packet16c>(const Packet16c& a) {
2763 Packet16c pair, quad, octo, result;
2764
2765 pair = vec_max(a, vec_sld(a, a, 8));
2766 quad = vec_max(pair, vec_sld(pair, pair, 4));
2767 octo = vec_max(quad, vec_sld(quad, quad, 2));
2768 result = vec_max(octo, vec_sld(octo, octo, 1));
2769
2770 return pfirst(result);
2771}
2772
2773template <>
2774EIGEN_STRONG_INLINE unsigned char predux_max<Packet16uc>(const Packet16uc& a) {
2775 Packet16uc pair, quad, octo, result;
2776
2777 pair = vec_max(a, vec_sld(a, a, 8));
2778 quad = vec_max(pair, vec_sld(pair, pair, 4));
2779 octo = vec_max(quad, vec_sld(quad, quad, 2));
2780 result = vec_max(octo, vec_sld(octo, octo, 1));
2781
2782 return pfirst(result);
2783}
2784
2785template <>
2786EIGEN_STRONG_INLINE bool predux_any(const Packet4f& x) {
2787 const Packet4ui zero = {0, 0, 0, 0};
2788 return vec_any_ne(reinterpret_cast<Packet4ui>(x), zero);
2789}
2790
2791template <>
2792EIGEN_STRONG_INLINE bool predux_any(const Packet8bf& x) {
2793 const Packet8us zero = {0, 0, 0, 0, 0, 0, 0, 0};
2794 return vec_any_ne(x.m_val, zero);
2795}
2796
2797template <typename T>
2798EIGEN_DEVICE_FUNC inline void ptranpose_common(PacketBlock<T, 4>& kernel) {
2799 T t0, t1, t2, t3;
2800 t0 = vec_mergeh(kernel.packet[0], kernel.packet[2]);
2801 t1 = vec_mergel(kernel.packet[0], kernel.packet[2]);
2802 t2 = vec_mergeh(kernel.packet[1], kernel.packet[3]);
2803 t3 = vec_mergel(kernel.packet[1], kernel.packet[3]);
2804 kernel.packet[0] = vec_mergeh(t0, t2);
2805 kernel.packet[1] = vec_mergel(t0, t2);
2806 kernel.packet[2] = vec_mergeh(t1, t3);
2807 kernel.packet[3] = vec_mergel(t1, t3);
2808}
2809
2810EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock<Packet4f, 4>& kernel) { ptranpose_common<Packet4f>(kernel); }
2811
2812EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock<Packet4i, 4>& kernel) { ptranpose_common<Packet4i>(kernel); }
2813
2814EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock<Packet8s, 4>& kernel) {
2815 Packet8s t0, t1, t2, t3;
2816 t0 = vec_mergeh(kernel.packet[0], kernel.packet[2]);
2817 t1 = vec_mergel(kernel.packet[0], kernel.packet[2]);
2818 t2 = vec_mergeh(kernel.packet[1], kernel.packet[3]);
2819 t3 = vec_mergel(kernel.packet[1], kernel.packet[3]);
2820 kernel.packet[0] = vec_mergeh(t0, t2);
2821 kernel.packet[1] = vec_mergel(t0, t2);
2822 kernel.packet[2] = vec_mergeh(t1, t3);
2823 kernel.packet[3] = vec_mergel(t1, t3);
2824}
2825
2826EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock<Packet8us, 4>& kernel) {
2827 Packet8us t0, t1, t2, t3;
2828 t0 = vec_mergeh(kernel.packet[0], kernel.packet[2]);
2829 t1 = vec_mergel(kernel.packet[0], kernel.packet[2]);
2830 t2 = vec_mergeh(kernel.packet[1], kernel.packet[3]);
2831 t3 = vec_mergel(kernel.packet[1], kernel.packet[3]);
2832 kernel.packet[0] = vec_mergeh(t0, t2);
2833 kernel.packet[1] = vec_mergel(t0, t2);
2834 kernel.packet[2] = vec_mergeh(t1, t3);
2835 kernel.packet[3] = vec_mergel(t1, t3);
2836}
2837
2838EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock<Packet8bf, 4>& kernel) {
2839 Packet8us t0, t1, t2, t3;
2840
2841 t0 = vec_mergeh(kernel.packet[0].m_val, kernel.packet[2].m_val);
2842 t1 = vec_mergel(kernel.packet[0].m_val, kernel.packet[2].m_val);
2843 t2 = vec_mergeh(kernel.packet[1].m_val, kernel.packet[3].m_val);
2844 t3 = vec_mergel(kernel.packet[1].m_val, kernel.packet[3].m_val);
2845 kernel.packet[0] = vec_mergeh(t0, t2);
2846 kernel.packet[1] = vec_mergel(t0, t2);
2847 kernel.packet[2] = vec_mergeh(t1, t3);
2848 kernel.packet[3] = vec_mergel(t1, t3);
2849}
2850
2851EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock<Packet16c, 4>& kernel) {
2852 Packet16c t0, t1, t2, t3;
2853 t0 = vec_mergeh(kernel.packet[0], kernel.packet[2]);
2854 t1 = vec_mergel(kernel.packet[0], kernel.packet[2]);
2855 t2 = vec_mergeh(kernel.packet[1], kernel.packet[3]);
2856 t3 = vec_mergel(kernel.packet[1], kernel.packet[3]);
2857 kernel.packet[0] = vec_mergeh(t0, t2);
2858 kernel.packet[1] = vec_mergel(t0, t2);
2859 kernel.packet[2] = vec_mergeh(t1, t3);
2860 kernel.packet[3] = vec_mergel(t1, t3);
2861}
2862
2863EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock<Packet16uc, 4>& kernel) {
2864 Packet16uc t0, t1, t2, t3;
2865 t0 = vec_mergeh(kernel.packet[0], kernel.packet[2]);
2866 t1 = vec_mergel(kernel.packet[0], kernel.packet[2]);
2867 t2 = vec_mergeh(kernel.packet[1], kernel.packet[3]);
2868 t3 = vec_mergel(kernel.packet[1], kernel.packet[3]);
2869 kernel.packet[0] = vec_mergeh(t0, t2);
2870 kernel.packet[1] = vec_mergel(t0, t2);
2871 kernel.packet[2] = vec_mergeh(t1, t3);
2872 kernel.packet[3] = vec_mergel(t1, t3);
2873}
2874
2875EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock<Packet8s, 8>& kernel) {
2876 Packet8s v[8], sum[8];
2877
2878 v[0] = vec_mergeh(kernel.packet[0], kernel.packet[4]);
2879 v[1] = vec_mergel(kernel.packet[0], kernel.packet[4]);
2880 v[2] = vec_mergeh(kernel.packet[1], kernel.packet[5]);
2881 v[3] = vec_mergel(kernel.packet[1], kernel.packet[5]);
2882 v[4] = vec_mergeh(kernel.packet[2], kernel.packet[6]);
2883 v[5] = vec_mergel(kernel.packet[2], kernel.packet[6]);
2884 v[6] = vec_mergeh(kernel.packet[3], kernel.packet[7]);
2885 v[7] = vec_mergel(kernel.packet[3], kernel.packet[7]);
2886 sum[0] = vec_mergeh(v[0], v[4]);
2887 sum[1] = vec_mergel(v[0], v[4]);
2888 sum[2] = vec_mergeh(v[1], v[5]);
2889 sum[3] = vec_mergel(v[1], v[5]);
2890 sum[4] = vec_mergeh(v[2], v[6]);
2891 sum[5] = vec_mergel(v[2], v[6]);
2892 sum[6] = vec_mergeh(v[3], v[7]);
2893 sum[7] = vec_mergel(v[3], v[7]);
2894
2895 kernel.packet[0] = vec_mergeh(sum[0], sum[4]);
2896 kernel.packet[1] = vec_mergel(sum[0], sum[4]);
2897 kernel.packet[2] = vec_mergeh(sum[1], sum[5]);
2898 kernel.packet[3] = vec_mergel(sum[1], sum[5]);
2899 kernel.packet[4] = vec_mergeh(sum[2], sum[6]);
2900 kernel.packet[5] = vec_mergel(sum[2], sum[6]);
2901 kernel.packet[6] = vec_mergeh(sum[3], sum[7]);
2902 kernel.packet[7] = vec_mergel(sum[3], sum[7]);
2903}
2904
2905EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock<Packet8us, 8>& kernel) {
2906 Packet8us v[8], sum[8];
2907
2908 v[0] = vec_mergeh(kernel.packet[0], kernel.packet[4]);
2909 v[1] = vec_mergel(kernel.packet[0], kernel.packet[4]);
2910 v[2] = vec_mergeh(kernel.packet[1], kernel.packet[5]);
2911 v[3] = vec_mergel(kernel.packet[1], kernel.packet[5]);
2912 v[4] = vec_mergeh(kernel.packet[2], kernel.packet[6]);
2913 v[5] = vec_mergel(kernel.packet[2], kernel.packet[6]);
2914 v[6] = vec_mergeh(kernel.packet[3], kernel.packet[7]);
2915 v[7] = vec_mergel(kernel.packet[3], kernel.packet[7]);
2916 sum[0] = vec_mergeh(v[0], v[4]);
2917 sum[1] = vec_mergel(v[0], v[4]);
2918 sum[2] = vec_mergeh(v[1], v[5]);
2919 sum[3] = vec_mergel(v[1], v[5]);
2920 sum[4] = vec_mergeh(v[2], v[6]);
2921 sum[5] = vec_mergel(v[2], v[6]);
2922 sum[6] = vec_mergeh(v[3], v[7]);
2923 sum[7] = vec_mergel(v[3], v[7]);
2924
2925 kernel.packet[0] = vec_mergeh(sum[0], sum[4]);
2926 kernel.packet[1] = vec_mergel(sum[0], sum[4]);
2927 kernel.packet[2] = vec_mergeh(sum[1], sum[5]);
2928 kernel.packet[3] = vec_mergel(sum[1], sum[5]);
2929 kernel.packet[4] = vec_mergeh(sum[2], sum[6]);
2930 kernel.packet[5] = vec_mergel(sum[2], sum[6]);
2931 kernel.packet[6] = vec_mergeh(sum[3], sum[7]);
2932 kernel.packet[7] = vec_mergel(sum[3], sum[7]);
2933}
2934
2935EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock<Packet8bf, 8>& kernel) {
2936 Packet8bf v[8], sum[8];
2937
2938 v[0] = vec_mergeh(kernel.packet[0].m_val, kernel.packet[4].m_val);
2939 v[1] = vec_mergel(kernel.packet[0].m_val, kernel.packet[4].m_val);
2940 v[2] = vec_mergeh(kernel.packet[1].m_val, kernel.packet[5].m_val);
2941 v[3] = vec_mergel(kernel.packet[1].m_val, kernel.packet[5].m_val);
2942 v[4] = vec_mergeh(kernel.packet[2].m_val, kernel.packet[6].m_val);
2943 v[5] = vec_mergel(kernel.packet[2].m_val, kernel.packet[6].m_val);
2944 v[6] = vec_mergeh(kernel.packet[3].m_val, kernel.packet[7].m_val);
2945 v[7] = vec_mergel(kernel.packet[3].m_val, kernel.packet[7].m_val);
2946 sum[0] = vec_mergeh(v[0].m_val, v[4].m_val);
2947 sum[1] = vec_mergel(v[0].m_val, v[4].m_val);
2948 sum[2] = vec_mergeh(v[1].m_val, v[5].m_val);
2949 sum[3] = vec_mergel(v[1].m_val, v[5].m_val);
2950 sum[4] = vec_mergeh(v[2].m_val, v[6].m_val);
2951 sum[5] = vec_mergel(v[2].m_val, v[6].m_val);
2952 sum[6] = vec_mergeh(v[3].m_val, v[7].m_val);
2953 sum[7] = vec_mergel(v[3].m_val, v[7].m_val);
2954
2955 kernel.packet[0] = vec_mergeh(sum[0].m_val, sum[4].m_val);
2956 kernel.packet[1] = vec_mergel(sum[0].m_val, sum[4].m_val);
2957 kernel.packet[2] = vec_mergeh(sum[1].m_val, sum[5].m_val);
2958 kernel.packet[3] = vec_mergel(sum[1].m_val, sum[5].m_val);
2959 kernel.packet[4] = vec_mergeh(sum[2].m_val, sum[6].m_val);
2960 kernel.packet[5] = vec_mergel(sum[2].m_val, sum[6].m_val);
2961 kernel.packet[6] = vec_mergeh(sum[3].m_val, sum[7].m_val);
2962 kernel.packet[7] = vec_mergel(sum[3].m_val, sum[7].m_val);
2963}
2964
2965EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock<Packet16c, 16>& kernel) {
2966 Packet16c step1[16], step2[16], step3[16];
2967
2968 step1[0] = vec_mergeh(kernel.packet[0], kernel.packet[8]);
2969 step1[1] = vec_mergel(kernel.packet[0], kernel.packet[8]);
2970 step1[2] = vec_mergeh(kernel.packet[1], kernel.packet[9]);
2971 step1[3] = vec_mergel(kernel.packet[1], kernel.packet[9]);
2972 step1[4] = vec_mergeh(kernel.packet[2], kernel.packet[10]);
2973 step1[5] = vec_mergel(kernel.packet[2], kernel.packet[10]);
2974 step1[6] = vec_mergeh(kernel.packet[3], kernel.packet[11]);
2975 step1[7] = vec_mergel(kernel.packet[3], kernel.packet[11]);
2976 step1[8] = vec_mergeh(kernel.packet[4], kernel.packet[12]);
2977 step1[9] = vec_mergel(kernel.packet[4], kernel.packet[12]);
2978 step1[10] = vec_mergeh(kernel.packet[5], kernel.packet[13]);
2979 step1[11] = vec_mergel(kernel.packet[5], kernel.packet[13]);
2980 step1[12] = vec_mergeh(kernel.packet[6], kernel.packet[14]);
2981 step1[13] = vec_mergel(kernel.packet[6], kernel.packet[14]);
2982 step1[14] = vec_mergeh(kernel.packet[7], kernel.packet[15]);
2983 step1[15] = vec_mergel(kernel.packet[7], kernel.packet[15]);
2984
2985 step2[0] = vec_mergeh(step1[0], step1[8]);
2986 step2[1] = vec_mergel(step1[0], step1[8]);
2987 step2[2] = vec_mergeh(step1[1], step1[9]);
2988 step2[3] = vec_mergel(step1[1], step1[9]);
2989 step2[4] = vec_mergeh(step1[2], step1[10]);
2990 step2[5] = vec_mergel(step1[2], step1[10]);
2991 step2[6] = vec_mergeh(step1[3], step1[11]);
2992 step2[7] = vec_mergel(step1[3], step1[11]);
2993 step2[8] = vec_mergeh(step1[4], step1[12]);
2994 step2[9] = vec_mergel(step1[4], step1[12]);
2995 step2[10] = vec_mergeh(step1[5], step1[13]);
2996 step2[11] = vec_mergel(step1[5], step1[13]);
2997 step2[12] = vec_mergeh(step1[6], step1[14]);
2998 step2[13] = vec_mergel(step1[6], step1[14]);
2999 step2[14] = vec_mergeh(step1[7], step1[15]);
3000 step2[15] = vec_mergel(step1[7], step1[15]);
3001
3002 step3[0] = vec_mergeh(step2[0], step2[8]);
3003 step3[1] = vec_mergel(step2[0], step2[8]);
3004 step3[2] = vec_mergeh(step2[1], step2[9]);
3005 step3[3] = vec_mergel(step2[1], step2[9]);
3006 step3[4] = vec_mergeh(step2[2], step2[10]);
3007 step3[5] = vec_mergel(step2[2], step2[10]);
3008 step3[6] = vec_mergeh(step2[3], step2[11]);
3009 step3[7] = vec_mergel(step2[3], step2[11]);
3010 step3[8] = vec_mergeh(step2[4], step2[12]);
3011 step3[9] = vec_mergel(step2[4], step2[12]);
3012 step3[10] = vec_mergeh(step2[5], step2[13]);
3013 step3[11] = vec_mergel(step2[5], step2[13]);
3014 step3[12] = vec_mergeh(step2[6], step2[14]);
3015 step3[13] = vec_mergel(step2[6], step2[14]);
3016 step3[14] = vec_mergeh(step2[7], step2[15]);
3017 step3[15] = vec_mergel(step2[7], step2[15]);
3018
3019 kernel.packet[0] = vec_mergeh(step3[0], step3[8]);
3020 kernel.packet[1] = vec_mergel(step3[0], step3[8]);
3021 kernel.packet[2] = vec_mergeh(step3[1], step3[9]);
3022 kernel.packet[3] = vec_mergel(step3[1], step3[9]);
3023 kernel.packet[4] = vec_mergeh(step3[2], step3[10]);
3024 kernel.packet[5] = vec_mergel(step3[2], step3[10]);
3025 kernel.packet[6] = vec_mergeh(step3[3], step3[11]);
3026 kernel.packet[7] = vec_mergel(step3[3], step3[11]);
3027 kernel.packet[8] = vec_mergeh(step3[4], step3[12]);
3028 kernel.packet[9] = vec_mergel(step3[4], step3[12]);
3029 kernel.packet[10] = vec_mergeh(step3[5], step3[13]);
3030 kernel.packet[11] = vec_mergel(step3[5], step3[13]);
3031 kernel.packet[12] = vec_mergeh(step3[6], step3[14]);
3032 kernel.packet[13] = vec_mergel(step3[6], step3[14]);
3033 kernel.packet[14] = vec_mergeh(step3[7], step3[15]);
3034 kernel.packet[15] = vec_mergel(step3[7], step3[15]);
3035}
3036
3037EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock<Packet16uc, 16>& kernel) {
3038 Packet16uc step1[16], step2[16], step3[16];
3039
3040 step1[0] = vec_mergeh(kernel.packet[0], kernel.packet[8]);
3041 step1[1] = vec_mergel(kernel.packet[0], kernel.packet[8]);
3042 step1[2] = vec_mergeh(kernel.packet[1], kernel.packet[9]);
3043 step1[3] = vec_mergel(kernel.packet[1], kernel.packet[9]);
3044 step1[4] = vec_mergeh(kernel.packet[2], kernel.packet[10]);
3045 step1[5] = vec_mergel(kernel.packet[2], kernel.packet[10]);
3046 step1[6] = vec_mergeh(kernel.packet[3], kernel.packet[11]);
3047 step1[7] = vec_mergel(kernel.packet[3], kernel.packet[11]);
3048 step1[8] = vec_mergeh(kernel.packet[4], kernel.packet[12]);
3049 step1[9] = vec_mergel(kernel.packet[4], kernel.packet[12]);
3050 step1[10] = vec_mergeh(kernel.packet[5], kernel.packet[13]);
3051 step1[11] = vec_mergel(kernel.packet[5], kernel.packet[13]);
3052 step1[12] = vec_mergeh(kernel.packet[6], kernel.packet[14]);
3053 step1[13] = vec_mergel(kernel.packet[6], kernel.packet[14]);
3054 step1[14] = vec_mergeh(kernel.packet[7], kernel.packet[15]);
3055 step1[15] = vec_mergel(kernel.packet[7], kernel.packet[15]);
3056
3057 step2[0] = vec_mergeh(step1[0], step1[8]);
3058 step2[1] = vec_mergel(step1[0], step1[8]);
3059 step2[2] = vec_mergeh(step1[1], step1[9]);
3060 step2[3] = vec_mergel(step1[1], step1[9]);
3061 step2[4] = vec_mergeh(step1[2], step1[10]);
3062 step2[5] = vec_mergel(step1[2], step1[10]);
3063 step2[6] = vec_mergeh(step1[3], step1[11]);
3064 step2[7] = vec_mergel(step1[3], step1[11]);
3065 step2[8] = vec_mergeh(step1[4], step1[12]);
3066 step2[9] = vec_mergel(step1[4], step1[12]);
3067 step2[10] = vec_mergeh(step1[5], step1[13]);
3068 step2[11] = vec_mergel(step1[5], step1[13]);
3069 step2[12] = vec_mergeh(step1[6], step1[14]);
3070 step2[13] = vec_mergel(step1[6], step1[14]);
3071 step2[14] = vec_mergeh(step1[7], step1[15]);
3072 step2[15] = vec_mergel(step1[7], step1[15]);
3073
3074 step3[0] = vec_mergeh(step2[0], step2[8]);
3075 step3[1] = vec_mergel(step2[0], step2[8]);
3076 step3[2] = vec_mergeh(step2[1], step2[9]);
3077 step3[3] = vec_mergel(step2[1], step2[9]);
3078 step3[4] = vec_mergeh(step2[2], step2[10]);
3079 step3[5] = vec_mergel(step2[2], step2[10]);
3080 step3[6] = vec_mergeh(step2[3], step2[11]);
3081 step3[7] = vec_mergel(step2[3], step2[11]);
3082 step3[8] = vec_mergeh(step2[4], step2[12]);
3083 step3[9] = vec_mergel(step2[4], step2[12]);
3084 step3[10] = vec_mergeh(step2[5], step2[13]);
3085 step3[11] = vec_mergel(step2[5], step2[13]);
3086 step3[12] = vec_mergeh(step2[6], step2[14]);
3087 step3[13] = vec_mergel(step2[6], step2[14]);
3088 step3[14] = vec_mergeh(step2[7], step2[15]);
3089 step3[15] = vec_mergel(step2[7], step2[15]);
3090
3091 kernel.packet[0] = vec_mergeh(step3[0], step3[8]);
3092 kernel.packet[1] = vec_mergel(step3[0], step3[8]);
3093 kernel.packet[2] = vec_mergeh(step3[1], step3[9]);
3094 kernel.packet[3] = vec_mergel(step3[1], step3[9]);
3095 kernel.packet[4] = vec_mergeh(step3[2], step3[10]);
3096 kernel.packet[5] = vec_mergel(step3[2], step3[10]);
3097 kernel.packet[6] = vec_mergeh(step3[3], step3[11]);
3098 kernel.packet[7] = vec_mergel(step3[3], step3[11]);
3099 kernel.packet[8] = vec_mergeh(step3[4], step3[12]);
3100 kernel.packet[9] = vec_mergel(step3[4], step3[12]);
3101 kernel.packet[10] = vec_mergeh(step3[5], step3[13]);
3102 kernel.packet[11] = vec_mergel(step3[5], step3[13]);
3103 kernel.packet[12] = vec_mergeh(step3[6], step3[14]);
3104 kernel.packet[13] = vec_mergel(step3[6], step3[14]);
3105 kernel.packet[14] = vec_mergeh(step3[7], step3[15]);
3106 kernel.packet[15] = vec_mergel(step3[7], step3[15]);
3107}
3108
3109//---------- double ----------
3110#ifdef EIGEN_VECTORIZE_VSX
3111typedef __vector double Packet2d;
3112typedef __vector unsigned long long Packet2ul;
3113typedef __vector long long Packet2l;
3114#if EIGEN_COMP_CLANG
3115typedef Packet2ul Packet2bl;
3116#else
3117typedef __vector __bool long Packet2bl;
3118#endif
3119
3120static Packet2l p2l_ZERO = reinterpret_cast<Packet2l>(p4i_ZERO);
3121static Packet2ul p2ul_SIGN = {0x8000000000000000ull, 0x8000000000000000ull};
3122static Packet2ul p2ul_PREV0DOT5 = {0x3FDFFFFFFFFFFFFFull, 0x3FDFFFFFFFFFFFFFull};
3123static Packet2d p2d_ONE = {1.0, 1.0};
3124static Packet2d p2d_ZERO = reinterpret_cast<Packet2d>(p4f_ZERO);
3125static Packet2d p2d_MZERO = {numext::bit_cast<double>(0x8000000000000000ull),
3126 numext::bit_cast<double>(0x8000000000000000ull)};
3127
3128#ifdef _BIG_ENDIAN
3129static Packet2d p2d_COUNTDOWN =
3130 reinterpret_cast<Packet2d>(vec_sld(reinterpret_cast<Packet4f>(p2d_ZERO), reinterpret_cast<Packet4f>(p2d_ONE), 8));
3131#else
3132static Packet2d p2d_COUNTDOWN =
3133 reinterpret_cast<Packet2d>(vec_sld(reinterpret_cast<Packet4f>(p2d_ONE), reinterpret_cast<Packet4f>(p2d_ZERO), 8));
3134#endif
3135
3136template <int index>
3137Packet2d vec_splat_dbl(const Packet2d& a) {
3138 return vec_splat(a, index);
3139}
3140
3141template <>
3142struct packet_traits<double> : default_packet_traits {
3143 typedef Packet2d type;
3144 typedef Packet2d half;
3145 enum {
3146 Vectorizable = 1,
3147 AlignedOnScalar = 1,
3148 size = 2,
3149
3150 HasAdd = 1,
3151 HasSub = 1,
3152 HasMul = 1,
3153 HasDiv = 1,
3154 HasMin = 1,
3155 HasMax = 1,
3156 HasAbs = 1,
3157 HasSin = EIGEN_FAST_MATH,
3158 HasCos = EIGEN_FAST_MATH,
3159 HasTan = EIGEN_FAST_MATH,
3160 HasTanh = EIGEN_FAST_MATH,
3161 HasErf = EIGEN_FAST_MATH,
3162 HasErfc = EIGEN_FAST_MATH,
3163 HasATanh = 1,
3164 HasATan = 0,
3165 HasCmp = 1,
3166 HasLog = 1,
3167 HasExp = 1,
3168 HasLog1p = 1,
3169 HasExpm1 = 1,
3170 HasSqrt = 1,
3171 HasCbrt = 1,
3172#if !EIGEN_COMP_CLANG
3173 HasRsqrt = 1,
3174#else
3175 HasRsqrt = 0,
3176#endif
3177 HasNegate = 1,
3178 };
3179};
3180
3181template <>
3182struct unpacket_traits<Packet2d> {
3183 typedef double type;
3184 typedef Packet2l integer_packet;
3185 enum {
3186 size = 2,
3187 alignment = Aligned16,
3188 vectorizable = true,
3189 masked_load_available = false,
3190 masked_store_available = false
3191 };
3192 typedef Packet2d half;
3193};
3194template <>
3195struct unpacket_traits<Packet2l> {
3196 typedef int64_t type;
3197 typedef Packet2l half;
3198 enum {
3199 size = 2,
3200 alignment = Aligned16,
3201 vectorizable = false,
3202 masked_load_available = false,
3203 masked_store_available = false
3204 };
3205};
3206
3207inline std::ostream& operator<<(std::ostream& s, const Packet2l& v) {
3208 union {
3209 Packet2l v;
3210 int64_t n[2];
3211 } vt;
3212 vt.v = v;
3213 s << vt.n[0] << ", " << vt.n[1];
3214 return s;
3215}
3216
3217inline std::ostream& operator<<(std::ostream& s, const Packet2d& v) {
3218 union {
3219 Packet2d v;
3220 double n[2];
3221 } vt;
3222 vt.v = v;
3223 s << vt.n[0] << ", " << vt.n[1];
3224 return s;
3225}
3226
3227// Need to define them first or we get specialization after instantiation errors
3228template <>
3229EIGEN_STRONG_INLINE Packet2d pload<Packet2d>(const double* from) {
3230 EIGEN_DEBUG_ALIGNED_LOAD
3231 return vec_xl(0, const_cast<double*>(from)); // cast needed by Clang
3232}
3233
3234template <>
3235EIGEN_ALWAYS_INLINE Packet2d pload_partial<Packet2d>(const double* from, const Index n, const Index offset) {
3236 return pload_partial_common<Packet2d>(from, n, offset);
3237}
3238
3239template <>
3240EIGEN_STRONG_INLINE void pstore<double>(double* to, const Packet2d& from) {
3241 EIGEN_DEBUG_ALIGNED_STORE
3242 vec_xst(from, 0, to);
3243}
3244
3245template <>
3246EIGEN_ALWAYS_INLINE void pstore_partial<double>(double* to, const Packet2d& from, const Index n, const Index offset) {
3247 pstore_partial_common<Packet2d>(to, from, n, offset);
3248}
3249
3250template <>
3251EIGEN_STRONG_INLINE Packet2d pset1<Packet2d>(const double& from) {
3252 Packet2d v = {from, from};
3253 return v;
3254}
3255template <>
3256EIGEN_STRONG_INLINE Packet2l pset1<Packet2l>(const int64_t& from) {
3257 Packet2l v = {from, from};
3258 return v;
3259}
3260
3261template <>
3262EIGEN_STRONG_INLINE Packet2d pset1frombits<Packet2d>(unsigned long from) {
3263 Packet2l v = {static_cast<long long>(from), static_cast<long long>(from)};
3264 return reinterpret_cast<Packet2d>(v);
3265}
3266
3267template <>
3268EIGEN_STRONG_INLINE void pbroadcast4<Packet2d>(const double* a, Packet2d& a0, Packet2d& a1, Packet2d& a2,
3269 Packet2d& a3) {
3270 // This way is faster than vec_splat (at least for doubles in Power 9)
3271 a0 = pset1<Packet2d>(a[0]);
3272 a1 = pset1<Packet2d>(a[1]);
3273 a2 = pset1<Packet2d>(a[2]);
3274 a3 = pset1<Packet2d>(a[3]);
3275}
3276
3277template <>
3278EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet2d pgather<double, Packet2d>(const double* from, Index stride) {
3279 return pgather_common<Packet2d>(from, stride);
3280}
3281template <>
3282EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE Packet2d pgather_partial<double, Packet2d>(const double* from, Index stride,
3283 const Index n) {
3284 return pgather_common<Packet2d>(from, stride, n);
3285}
3286template <>
3287EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void pscatter<double, Packet2d>(double* to, const Packet2d& from, Index stride) {
3288 pscatter_common<Packet2d>(to, from, stride);
3289}
3290template <>
3291EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void pscatter_partial<double, Packet2d>(double* to, const Packet2d& from,
3292 Index stride, const Index n) {
3293 pscatter_common<Packet2d>(to, from, stride, n);
3294}
3295
3296template <>
3297EIGEN_STRONG_INLINE Packet2d plset<Packet2d>(const double& a) {
3298 return pset1<Packet2d>(a) + p2d_COUNTDOWN;
3299}
3300
3301template <>
3302EIGEN_STRONG_INLINE Packet2d padd<Packet2d>(const Packet2d& a, const Packet2d& b) {
3303 return a + b;
3304}
3305
3306template <>
3307EIGEN_STRONG_INLINE Packet2d psub<Packet2d>(const Packet2d& a, const Packet2d& b) {
3308 return a - b;
3309}
3310
3311template <>
3312EIGEN_STRONG_INLINE Packet2d pnegate(const Packet2d& a) {
3313#ifdef EIGEN_VECTORIZE_POWER8_VECTOR
3314 return vec_neg(a);
3315#else
3316 return vec_xor(a, p2d_MZERO);
3317#endif
3318}
3319
3320template <>
3321EIGEN_STRONG_INLINE Packet2d pmul<Packet2d>(const Packet2d& a, const Packet2d& b) {
3322 return vec_madd(a, b, p2d_MZERO);
3323}
3324template <>
3325EIGEN_STRONG_INLINE Packet2d pdiv<Packet2d>(const Packet2d& a, const Packet2d& b) {
3326 return vec_div(a, b);
3327}
3328
3329// This overload is required for integer packet types.
3330template <>
3331EIGEN_STRONG_INLINE Packet2d pmadd(const Packet2d& a, const Packet2d& b, const Packet2d& c) {
3332 return vec_madd(a, b, c);
3333}
3334template <>
3335EIGEN_STRONG_INLINE Packet2d pmsub(const Packet2d& a, const Packet2d& b, const Packet2d& c) {
3336 return vec_msub(a, b, c);
3337}
3338template <>
3339EIGEN_STRONG_INLINE Packet2d pnmadd(const Packet2d& a, const Packet2d& b, const Packet2d& c) {
3340 return vec_nmsub(a, b, c);
3341}
3342template <>
3343EIGEN_STRONG_INLINE Packet2d pnmsub(const Packet2d& a, const Packet2d& b, const Packet2d& c) {
3344 return vec_nmadd(a, b, c);
3345}
3346
3347template <>
3348EIGEN_STRONG_INLINE Packet2d pmin<Packet2d>(const Packet2d& a, const Packet2d& b) {
3349 // NOTE: about 10% slower than vec_min, but consistent with std::min and SSE regarding NaN
3350 Packet2d ret;
3351 __asm__("xvcmpgedp %x0,%x1,%x2\n\txxsel %x0,%x1,%x2,%x0" : "=&wa"(ret) : "wa"(a), "wa"(b));
3352 return ret;
3353}
3354
3355template <>
3356EIGEN_STRONG_INLINE Packet2d pmax<Packet2d>(const Packet2d& a, const Packet2d& b) {
3357 // NOTE: about 10% slower than vec_max, but consistent with std::max and SSE regarding NaN
3358 Packet2d ret;
3359 __asm__("xvcmpgtdp %x0,%x2,%x1\n\txxsel %x0,%x1,%x2,%x0" : "=&wa"(ret) : "wa"(a), "wa"(b));
3360 return ret;
3361}
3362
3363template <>
3364EIGEN_STRONG_INLINE Packet2d pcmp_le(const Packet2d& a, const Packet2d& b) {
3365 return reinterpret_cast<Packet2d>(vec_cmple(a, b));
3366}
3367template <>
3368EIGEN_STRONG_INLINE Packet2d pcmp_lt(const Packet2d& a, const Packet2d& b) {
3369 return reinterpret_cast<Packet2d>(vec_cmplt(a, b));
3370}
3371template <>
3372EIGEN_STRONG_INLINE Packet2d pcmp_eq(const Packet2d& a, const Packet2d& b) {
3373 return reinterpret_cast<Packet2d>(vec_cmpeq(a, b));
3374}
3375template <>
3376#ifdef EIGEN_VECTORIZE_POWER8_VECTOR
3377EIGEN_STRONG_INLINE Packet2l pcmp_eq(const Packet2l& a, const Packet2l& b) {
3378 return reinterpret_cast<Packet2l>(vec_cmpeq(a, b));
3379}
3380#else
3381EIGEN_STRONG_INLINE Packet2l pcmp_eq(const Packet2l& a, const Packet2l& b) {
3382 Packet4i halves = reinterpret_cast<Packet4i>(vec_cmpeq(reinterpret_cast<Packet4i>(a), reinterpret_cast<Packet4i>(b)));
3383 Packet4i flipped = vec_perm(halves, halves, p16uc_COMPLEX32_REV);
3384 return reinterpret_cast<Packet2l>(pand(halves, flipped));
3385}
3386#endif
3387// Not the generic a < b ? ptrue(a) : pzero(a): under Clang's default -faltivec-src-compat=mixed,
3388// a < b on vector long long is a scalar all-lanes predicate, not a lane mask.
3389template <>
3390#ifdef EIGEN_VECTORIZE_POWER8_VECTOR
3391EIGEN_STRONG_INLINE Packet2l pcmp_lt(const Packet2l& a, const Packet2l& b) {
3392 return reinterpret_cast<Packet2l>(vec_cmplt(a, b));
3393}
3394#else
3395EIGEN_STRONG_INLINE Packet2l pcmp_lt(const Packet2l& a, const Packet2l& b) {
3396 const Packet2l ret = {a[0] < b[0] ? -1 : 0, a[1] < b[1] ? -1 : 0};
3397 return ret;
3398}
3399#endif
3400template <>
3401EIGEN_STRONG_INLINE Packet2d pcmp_lt_or_nan(const Packet2d& a, const Packet2d& b) {
3402 Packet2d c = reinterpret_cast<Packet2d>(vec_cmpge(a, b));
3403 return vec_nor(c, c);
3404}
3405
3406template <>
3407EIGEN_STRONG_INLINE Packet2d pand<Packet2d>(const Packet2d& a, const Packet2d& b) {
3408 return vec_and(a, b);
3409}
3410
3411template <>
3412EIGEN_STRONG_INLINE Packet2d por<Packet2d>(const Packet2d& a, const Packet2d& b) {
3413 return vec_or(a, b);
3414}
3415
3416template <>
3417EIGEN_STRONG_INLINE Packet2d pxor<Packet2d>(const Packet2d& a, const Packet2d& b) {
3418 return vec_xor(a, b);
3419}
3420
3421template <>
3422EIGEN_STRONG_INLINE Packet2d pandnot<Packet2d>(const Packet2d& a, const Packet2d& b) {
3423 return vec_and(a, vec_nor(b, b));
3424}
3425
3426template <>
3427EIGEN_STRONG_INLINE Packet2d pround<Packet2d>(const Packet2d& a) {
3428 Packet2d t = vec_add(
3429 reinterpret_cast<Packet2d>(vec_or(vec_and(reinterpret_cast<Packet2ul>(a), p2ul_SIGN), p2ul_PREV0DOT5)), a);
3430 Packet2d res;
3431
3432 __asm__("xvrdpiz %x0, %x1\n\t" : "=&wa"(res) : "wa"(t));
3433
3434 return res;
3435}
3436template <>
3437EIGEN_STRONG_INLINE Packet2d pceil<Packet2d>(const Packet2d& a) {
3438 return vec_ceil(a);
3439}
3440template <>
3441EIGEN_STRONG_INLINE Packet2d pfloor<Packet2d>(const Packet2d& a) {
3442 return vec_floor(a);
3443}
3444template <>
3445EIGEN_STRONG_INLINE Packet2d ptrunc<Packet2d>(const Packet2d& a) {
3446 return vec_trunc(a);
3447}
3448template <>
3449EIGEN_STRONG_INLINE Packet2d print<Packet2d>(const Packet2d& a) {
3450 Packet2d res;
3451
3452 __asm__("xvrdpic %x0, %x1\n\t" : "=&wa"(res) : "wa"(a));
3453
3454 return res;
3455}
3456
3457template <>
3458EIGEN_STRONG_INLINE Packet2d ploadu<Packet2d>(const double* from) {
3459 EIGEN_DEBUG_UNALIGNED_LOAD
3460 return vec_xl(0, const_cast<double*>(from));
3461}
3462
3463template <>
3464EIGEN_ALWAYS_INLINE Packet2d ploadu_partial<Packet2d>(const double* from, const Index n, const Index offset) {
3465 return ploadu_partial_common<Packet2d>(from, n, offset);
3466}
3467
3468template <>
3469EIGEN_STRONG_INLINE Packet2d ploaddup<Packet2d>(const double* from) {
3470 Packet2d p;
3471 if ((std::ptrdiff_t(from) % 16) == 0)
3472 p = pload<Packet2d>(from);
3473 else
3474 p = ploadu<Packet2d>(from);
3475 return vec_splat_dbl<0>(p);
3476}
3477
3478template <>
3479EIGEN_STRONG_INLINE void pstoreu<double>(double* to, const Packet2d& from) {
3480 EIGEN_DEBUG_UNALIGNED_STORE
3481 vec_xst(from, 0, to);
3482}
3483
3484template <>
3485EIGEN_ALWAYS_INLINE void pstoreu_partial<double>(double* to, const Packet2d& from, const Index n, const Index offset) {
3486 pstoreu_partial_common<Packet2d>(to, from, n, offset);
3487}
3488
3489template <>
3490EIGEN_STRONG_INLINE void prefetch<double>(const double* addr) {
3491 EIGEN_PPC_PREFETCH(addr);
3492}
3493
3494template <>
3495EIGEN_STRONG_INLINE double pfirst<Packet2d>(const Packet2d& a) {
3496 EIGEN_ALIGN16 double x[2];
3497 pstore<double>(x, a);
3498 return x[0];
3499}
3500
3501template <>
3502EIGEN_STRONG_INLINE Packet2d preverse(const Packet2d& a) {
3503 return vec_sld(a, a, 8);
3504}
3505template <>
3506EIGEN_STRONG_INLINE Packet2d pabs(const Packet2d& a) {
3507 return vec_abs(a);
3508}
3509#ifdef EIGEN_VECTORIZE_POWER8_VECTOR
3510template <>
3511EIGEN_STRONG_INLINE Packet2d psignbit(const Packet2d& a) {
3512 return (Packet2d)vec_sra((Packet2l)a, vec_splats((unsigned long long)(63)));
3513}
3514#else
3515#ifdef _BIG_ENDIAN
3516static Packet16uc p16uc_DUPSIGN = {0, 0, 0, 0, 0, 0, 0, 0, 8, 8, 8, 8, 8, 8, 8, 8};
3517#else
3518static Packet16uc p16uc_DUPSIGN = {7, 7, 7, 7, 7, 7, 7, 7, 15, 15, 15, 15, 15, 15, 15, 15};
3519#endif
3520
3521template <>
3522EIGEN_STRONG_INLINE Packet2d psignbit(const Packet2d& a) {
3523 Packet16c tmp = vec_sra(reinterpret_cast<Packet16c>(a), vec_splats((unsigned char)(7)));
3524 return reinterpret_cast<Packet2d>(vec_perm(tmp, tmp, p16uc_DUPSIGN));
3525}
3526#endif
3527
3528template <>
3529inline Packet2l pcast<Packet2d, Packet2l>(const Packet2d& x);
3530
3531template <>
3532inline Packet2d pcast<Packet2l, Packet2d>(const Packet2l& x);
3533
3534// Packet2l shifts.
3535// For POWER8 we simply use vec_sr/l.
3536//
3537// Things are more complicated for POWER7. There is actually a
3538// vec_xxsxdi intrinsic but it is not supported by some gcc versions.
3539// So we need to shift by N % 32 and rearrange bytes.
3540#ifdef EIGEN_VECTORIZE_POWER8_VECTOR
3541
3542template <int N>
3543EIGEN_STRONG_INLINE Packet2l plogical_shift_left(const Packet2l& a) {
3544 const Packet2ul shift = {N, N};
3545 return vec_sl(a, shift);
3546}
3547
3548template <int N>
3549EIGEN_STRONG_INLINE Packet2l plogical_shift_right(const Packet2l& a) {
3550 const Packet2ul shift = {N, N};
3551 return vec_sr(a, shift);
3552}
3553
3554#else
3555
3556// Shifts [A, B, C, D] to [B, 0, D, 0].
3557// Used to implement left shifts for Packet2l.
3558EIGEN_ALWAYS_INLINE Packet4i shift_even_left(const Packet4i& a) {
3559 static const Packet16uc perm = {0x14, 0x15, 0x16, 0x17, 0x00, 0x01, 0x02, 0x03,
3560 0x1c, 0x1d, 0x1e, 0x1f, 0x08, 0x09, 0x0a, 0x0b};
3561#ifdef _BIG_ENDIAN
3562 return vec_perm(p4i_ZERO, a, perm);
3563#else
3564 return vec_perm(a, p4i_ZERO, perm);
3565#endif
3566}
3567
3568// Shifts [A, B, C, D] to [0, A, 0, C].
3569// Used to implement right shifts for Packet2l.
3570EIGEN_ALWAYS_INLINE Packet4i shift_odd_right(const Packet4i& a) {
3571 static const Packet16uc perm = {0x04, 0x05, 0x06, 0x07, 0x10, 0x11, 0x12, 0x13,
3572 0x0c, 0x0d, 0x0e, 0x0f, 0x18, 0x19, 0x1a, 0x1b};
3573#ifdef _BIG_ENDIAN
3574 return vec_perm(p4i_ZERO, a, perm);
3575#else
3576 return vec_perm(a, p4i_ZERO, perm);
3577#endif
3578}
3579
3580template <int N, typename EnableIf = void>
3581struct plogical_shift_left_impl;
3582
3583template <int N>
3584struct plogical_shift_left_impl<N, std::enable_if_t<(N < 32) && (N >= 0)> > {
3585 static EIGEN_STRONG_INLINE Packet2l run(const Packet2l& a) {
3586 static const unsigned n = static_cast<unsigned>(N);
3587 const Packet4ui shift = {n, n, n, n};
3588 const Packet4i ai = reinterpret_cast<Packet4i>(a);
3589 static const unsigned m = static_cast<unsigned>(32 - N);
3590 const Packet4ui shift_right = {m, m, m, m};
3591 const Packet4i out_hi = vec_sl(ai, shift);
3592 const Packet4i out_lo = shift_even_left(vec_sr(ai, shift_right));
3593 return reinterpret_cast<Packet2l>(por<Packet4i>(out_hi, out_lo));
3594 }
3595};
3596
3597template <int N>
3598struct plogical_shift_left_impl<N, std::enable_if_t<(N >= 32)> > {
3599 static EIGEN_STRONG_INLINE Packet2l run(const Packet2l& a) {
3600 static const unsigned m = static_cast<unsigned>(N - 32);
3601 const Packet4ui shift = {m, m, m, m};
3602 const Packet4i ai = reinterpret_cast<Packet4i>(a);
3603 return reinterpret_cast<Packet2l>(shift_even_left(vec_sl(ai, shift)));
3604 }
3605};
3606
3607template <int N>
3608EIGEN_STRONG_INLINE Packet2l plogical_shift_left(const Packet2l& a) {
3609 return plogical_shift_left_impl<N>::run(a);
3610}
3611
3612template <int N, typename EnableIf = void>
3613struct plogical_shift_right_impl;
3614
3615template <int N>
3616struct plogical_shift_right_impl<N, std::enable_if_t<(N < 32) && (N >= 0)> > {
3617 static EIGEN_STRONG_INLINE Packet2l run(const Packet2l& a) {
3618 static const unsigned n = static_cast<unsigned>(N);
3619 const Packet4ui shift = {n, n, n, n};
3620 const Packet4i ai = reinterpret_cast<Packet4i>(a);
3621 static const unsigned m = static_cast<unsigned>(32 - N);
3622 const Packet4ui shift_left = {m, m, m, m};
3623 const Packet4i out_lo = vec_sr(ai, shift);
3624 const Packet4i out_hi = shift_odd_right(vec_sl(ai, shift_left));
3625 return reinterpret_cast<Packet2l>(por<Packet4i>(out_hi, out_lo));
3626 }
3627};
3628
3629template <int N>
3630struct plogical_shift_right_impl<N, std::enable_if_t<(N >= 32)> > {
3631 static EIGEN_STRONG_INLINE Packet2l run(const Packet2l& a) {
3632 static const unsigned m = static_cast<unsigned>(N - 32);
3633 const Packet4ui shift = {m, m, m, m};
3634 const Packet4i ai = reinterpret_cast<Packet4i>(a);
3635 return reinterpret_cast<Packet2l>(shift_odd_right(vec_sr(ai, shift)));
3636 }
3637};
3638
3639template <int N>
3640EIGEN_STRONG_INLINE Packet2l plogical_shift_right(const Packet2l& a) {
3641 return plogical_shift_right_impl<N>::run(a);
3642}
3643#endif
3644
3645template <>
3646EIGEN_STRONG_INLINE Packet2d pldexp<Packet2d>(const Packet2d& a, const Packet2d& exponent) {
3647 // The single-rounding split of pldexp_generic on int64 exponents.
3648 const Packet2d max_exponent = pset1<Packet2d>(2099.0);
3649 const Packet2d last_max = pset1<Packet2d>(1022.0);
3650 const Packet2l e = pcast<Packet2d, Packet2l>(pmin(pmax(exponent, pnegate(max_exponent)), max_exponent));
3651 const Packet2l b = pcast<Packet2d, Packet2l>(pmin(pmax(exponent, pnegate(last_max)), last_max));
3652 const Packet2l bias = {1023, 1023};
3653 const Packet2l even = {-2, -2};
3654 const Packet2l t = psub(e, b) & even;
3655 const Packet2d c1 = reinterpret_cast<Packet2d>(plogical_shift_left<51>(t + bias + bias)); // 2^(t/2)
3656 const Packet2d c2 = reinterpret_cast<Packet2d>(plogical_shift_left<52>(psub(e, t) + bias)); // 2^(e - t)
3657 return pldexp_apply_factors(a, c1, c2); // a * 2^e
3658}
3659
3660// Extract exponent without existence of Packet2l.
3661template <>
3662EIGEN_STRONG_INLINE Packet2d pfrexp_generic_get_biased_exponent(const Packet2d& a) {
3663 return pcast<Packet2l, Packet2d>(plogical_shift_right<52>(reinterpret_cast<Packet2l>(pabs(a))));
3664}
3665
3666template <>
3667EIGEN_STRONG_INLINE Packet2d pfrexp<Packet2d>(const Packet2d& a, Packet2d& exponent) {
3668 return pfrexp_generic(a, exponent);
3669}
3670
3671template <>
3672EIGEN_STRONG_INLINE double predux<Packet2d>(const Packet2d& a) {
3673 Packet2d b, sum;
3674 b = reinterpret_cast<Packet2d>(vec_sld(reinterpret_cast<Packet4f>(a), reinterpret_cast<Packet4f>(a), 8));
3675 sum = a + b;
3676 return pfirst<Packet2d>(sum);
3677}
3678
3679template <>
3680EIGEN_STRONG_INLINE bool predux_any(const Packet2d& a) {
3681 // A 64-bit lane is nonzero iff one of its 32-bit halves is; the doubleword compare needs POWER8.
3682 const Packet4ui zero = {0, 0, 0, 0};
3683 return vec_any_ne(reinterpret_cast<Packet4ui>(a), zero);
3684}
3685
3686// Other reduction functions:
3687// mul
3688template <>
3689EIGEN_STRONG_INLINE double predux_mul<Packet2d>(const Packet2d& a) {
3690 return pfirst(
3691 pmul(a, reinterpret_cast<Packet2d>(vec_sld(reinterpret_cast<Packet4ui>(a), reinterpret_cast<Packet4ui>(a), 8))));
3692}
3693
3694// min
3695template <>
3696EIGEN_STRONG_INLINE double predux_min<Packet2d>(const Packet2d& a) {
3697 return pfirst(
3698 pmin(a, reinterpret_cast<Packet2d>(vec_sld(reinterpret_cast<Packet4ui>(a), reinterpret_cast<Packet4ui>(a), 8))));
3699}
3700
3701// max
3702template <>
3703EIGEN_STRONG_INLINE double predux_max<Packet2d>(const Packet2d& a) {
3704 return pfirst(
3705 pmax(a, reinterpret_cast<Packet2d>(vec_sld(reinterpret_cast<Packet4ui>(a), reinterpret_cast<Packet4ui>(a), 8))));
3706}
3707
3708EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock<Packet2d, 2>& kernel) {
3709 Packet2d t0, t1;
3710 t0 = vec_mergeh(kernel.packet[0], kernel.packet[1]);
3711 t1 = vec_mergel(kernel.packet[0], kernel.packet[1]);
3712 kernel.packet[0] = t0;
3713 kernel.packet[1] = t1;
3714}
3715
3716#endif // __VSX__
3717} // end namespace internal
3718
3719} // end namespace Eigen
3720
3721#endif // EIGEN_PACKET_MATH_ALTIVEC_H
@ Aligned16
Definition Constants.h:238