Eigen  5.0.1
 
Loading...
Searching...
No Matches
PacketMath.h
1// This file is part of Eigen, a lightweight C++ template library
2// for linear algebra.
3//
4// Copyright (C) 2016 Konstantinos Margaritis <markos@freevec.org>
5//
6// This Source Code Form is subject to the terms of the Mozilla
7// Public License v. 2.0. If a copy of the MPL was not distributed
8// with this file, You can obtain one at http://mozilla.org/MPL/2.0/.
9// SPDX-License-Identifier: MPL-2.0
10
11#ifndef EIGEN_PACKET_MATH_ZVECTOR_H
12#define EIGEN_PACKET_MATH_ZVECTOR_H
13
14// IWYU pragma: private
15#include "../../InternalHeaderCheck.h"
16
17namespace Eigen {
18
19namespace internal {
20
21#ifndef EIGEN_CACHEFRIENDLY_PRODUCT_THRESHOLD
22#define EIGEN_CACHEFRIENDLY_PRODUCT_THRESHOLD 16
23#endif
24
25#ifndef EIGEN_HAS_SINGLE_INSTRUCTION_MADD
26#define EIGEN_HAS_SINGLE_INSTRUCTION_MADD
27#endif
28
29#ifndef EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS
30#define EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS 32
31#endif
32
33typedef __vector int Packet4i;
34typedef __vector unsigned int Packet4ui;
35typedef __vector __bool int Packet4bi;
36typedef __vector short int Packet8i;
37typedef __vector unsigned char Packet16uc;
38typedef __vector double Packet2d;
39typedef __vector unsigned long long Packet2ul;
40typedef __vector long long Packet2l;
41
42// Z14 has builtin support for float vectors
43typedef __vector float Packet4f;
44
45typedef union {
46 numext::int32_t i[4];
47 numext::uint32_t ui[4];
48 numext::int64_t l[2];
49 numext::uint64_t ul[2];
50 double d[2];
51 float f[4];
52 Packet4i v4i;
53 Packet4ui v4ui;
54 Packet2l v2l;
55 Packet2ul v2ul;
56 Packet2d v2d;
57 Packet4f v4f;
58} Packet;
59
60// We don't want to write the same code all the time, but we need to reuse the constants
61// and it doesn't really work to declare them global, so we define macros instead
62
63#define EIGEN_DECLARE_CONST_FAST_Packet4i(NAME, X) Packet4i p4i_##NAME = reinterpret_cast<Packet4i>(vec_splat_s32(X))
64
65#define EIGEN_DECLARE_CONST_FAST_Packet2d(NAME, X) Packet2d p2d_##NAME = reinterpret_cast<Packet2d>(vec_splat_s64(X))
66
67#define EIGEN_DECLARE_CONST_FAST_Packet2l(NAME, X) Packet2l p2l_##NAME = reinterpret_cast<Packet2l>(vec_splat_s64(X))
68
69#define EIGEN_DECLARE_CONST_Packet4i(NAME, X) Packet4i p4i_##NAME = pset1<Packet4i>(X)
70
71#define EIGEN_DECLARE_CONST_Packet2d(NAME, X) Packet2d p2d_##NAME = pset1<Packet2d>(X)
72
73#define EIGEN_DECLARE_CONST_Packet2l(NAME, X) Packet2l p2l_##NAME = pset1<Packet2l>(X)
74
75// These constants are endian-agnostic
76static EIGEN_DECLARE_CONST_FAST_Packet4i(ZERO, 0); //{ 0, 0, 0, 0,}
77static EIGEN_DECLARE_CONST_FAST_Packet4i(ONE, 1); //{ 1, 1, 1, 1}
78
79static EIGEN_DECLARE_CONST_FAST_Packet2d(ZERO, 0);
80static EIGEN_DECLARE_CONST_FAST_Packet2l(ZERO, 0);
81static EIGEN_DECLARE_CONST_FAST_Packet2l(ONE, 1);
82
83static Packet2d p2d_ONE = {1.0, 1.0};
84static Packet2d p2d_ZERO_ = {numext::bit_cast<double>(0x8000000000000000ull),
85 numext::bit_cast<double>(0x8000000000000000ull)};
86
87#define EIGEN_DECLARE_CONST_FAST_Packet4f(NAME, X) Packet4f p4f_##NAME = reinterpret_cast<Packet4f>(vec_splat_s32(X))
88
89#define EIGEN_DECLARE_CONST_Packet4f(NAME, X) Packet4f p4f_##NAME = pset1<Packet4f>(X)
90
91#define EIGEN_DECLARE_CONST_Packet4f_FROM_INT(NAME, X) \
92 const Packet4f p4f_##NAME = reinterpret_cast<Packet4f>(pset1<Packet4i>(X))
93
94static EIGEN_DECLARE_CONST_FAST_Packet4f(ZERO, 0); //{ 0.0, 0.0, 0.0, 0.0}
95static EIGEN_DECLARE_CONST_FAST_Packet4i(MINUS1, -1); //{ -1, -1, -1, -1}
96static Packet4f p4f_MZERO = {0x80000000, 0x80000000, 0x80000000, 0x80000000};
97
98static Packet4i p4i_COUNTDOWN = {0, 1, 2, 3};
99static Packet4f p4f_COUNTDOWN = {0.0, 1.0, 2.0, 3.0};
100static Packet2d p2d_COUNTDOWN = reinterpret_cast<Packet2d>(
101 vec_sld(reinterpret_cast<Packet16uc>(p2d_ZERO), reinterpret_cast<Packet16uc>(p2d_ONE), 8));
102
103static Packet16uc p16uc_PSET64_HI = {0, 1, 2, 3, 4, 5, 6, 7, 0, 1, 2, 3, 4, 5, 6, 7};
104static Packet16uc p16uc_DUPLICATE32_HI = {0, 1, 2, 3, 0, 1, 2, 3, 4, 5, 6, 7, 4, 5, 6, 7};
105
106// Mask alignment
107#define EIGEN_MASK_ALIGNMENT 0xfffffffffffffff0
108
109#define EIGEN_ALIGNED_PTR(x) ((std::ptrdiff_t)(x) & EIGEN_MASK_ALIGNMENT)
110
111// Handle endianness properly while loading constants
112// Define global static constants:
113
114static Packet16uc p16uc_FORWARD = {0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15};
115static Packet16uc p16uc_REVERSE32 = {12, 13, 14, 15, 8, 9, 10, 11, 4, 5, 6, 7, 0, 1, 2, 3};
116static Packet16uc p16uc_REVERSE64 = {8, 9, 10, 11, 12, 13, 14, 15, 0, 1, 2, 3, 4, 5, 6, 7};
117
118static Packet16uc p16uc_PSET32_WODD =
119 vec_sld((Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 0), (Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 2),
120 8); //{ 0,1,2,3, 0,1,2,3, 8,9,10,11, 8,9,10,11 };
121static Packet16uc p16uc_PSET32_WEVEN = vec_sld(p16uc_DUPLICATE32_HI, (Packet16uc)vec_splat((Packet4ui)p16uc_FORWARD, 3),
122 8); //{ 4,5,6,7, 4,5,6,7, 12,13,14,15, 12,13,14,15 };
123static Packet16uc p16uc_PSET64_LO = (Packet16uc)vec_mergel(
124 (Packet4ui)p16uc_PSET32_WODD, (Packet4ui)p16uc_PSET32_WEVEN); //{ 8,9,10,11, 12,13,14,15, 8,9,10,11, 12,13,14,15 };
125static Packet16uc p16uc_TRANSPOSE64_HI = {0, 1, 2, 3, 4, 5, 6, 7, 16, 17, 18, 19, 20, 21, 22, 23};
126static Packet16uc p16uc_TRANSPOSE64_LO = {8, 9, 10, 11, 12, 13, 14, 15, 24, 25, 26, 27, 28, 29, 30, 31};
127
128static Packet16uc p16uc_COMPLEX32_REV =
129 vec_sld(p16uc_REVERSE32, p16uc_REVERSE32, 8); //{ 4,5,6,7, 0,1,2,3, 12,13,14,15, 8,9,10,11 };
130
131static Packet16uc p16uc_COMPLEX32_REV2 =
132 vec_sld(p16uc_FORWARD, p16uc_FORWARD, 8); //{ 8,9,10,11, 12,13,14,15, 0,1,2,3, 4,5,6,7 };
133
134#if EIGEN_HAS_BUILTIN(__builtin_prefetch) || EIGEN_COMP_GNUC
135#define EIGEN_ZVECTOR_PREFETCH(ADDR) __builtin_prefetch(ADDR);
136#else
137#define EIGEN_ZVECTOR_PREFETCH(ADDR) asm(" pfd [%[addr]]\n" ::[addr] "r"(ADDR) : "cc");
138#endif
139
140template <>
141struct packet_traits<int> : default_packet_traits {
142 typedef Packet4i type;
143 typedef Packet4i half;
144 enum {
145 Vectorizable = 1,
146 AlignedOnScalar = 1,
147 size = 4,
148
149 HasAdd = 1,
150 HasSub = 1,
151 HasMul = 1,
152 HasDiv = 1,
153 };
154};
155
156template <>
157struct packet_traits<float> : default_packet_traits {
158 typedef Packet4f type;
159 typedef Packet4f half;
160 enum {
161 Vectorizable = 1,
162 AlignedOnScalar = 1,
163 size = 4,
164
165 HasCmp = 1,
166 HasAdd = 1,
167 HasSub = 1,
168 HasMul = 1,
169 HasDiv = 1,
170 HasMin = 1,
171 HasMax = 1,
172 HasAbs = 1,
173 HasSin = 0,
174 HasCos = 0,
175 HasLog = 0,
176 HasExp = 1,
177 HasSqrt = 1,
178 HasRsqrt = 1,
179 HasTanh = 1,
180 HasErf = 1,
181 HasNegate = 1,
182 };
183};
184
185template <>
186struct packet_traits<double> : default_packet_traits {
187 typedef Packet2d type;
188 typedef Packet2d half;
189 enum {
190 Vectorizable = 1,
191 AlignedOnScalar = 1,
192 size = 2,
193
194 HasAdd = 1,
195 HasSub = 1,
196 HasMul = 1,
197 HasDiv = 1,
198 HasMin = 1,
199 HasMax = 1,
200 HasAbs = 1,
201 HasSin = 0,
202 HasCos = 0,
203 HasLog = 0,
204 HasExp = 1,
205 HasSqrt = 1,
206 HasRsqrt = 1,
207 HasNegate = 1,
208 };
209};
210
211template <>
212struct unpacket_traits<Packet4i> {
213 typedef int type;
214 enum {
215 size = 4,
216 alignment = Aligned16,
217 vectorizable = true,
218 masked_load_available = false,
219 masked_store_available = false
220 };
221 typedef Packet4i half;
222};
223template <>
224struct unpacket_traits<Packet4f> {
225 typedef float type;
226 enum {
227 size = 4,
228 alignment = Aligned16,
229 vectorizable = true,
230 masked_load_available = false,
231 masked_store_available = false
232 };
233 typedef Packet4f half;
234 typedef Packet4i integer_packet;
235};
236template <>
237struct unpacket_traits<Packet2d> {
238 typedef double type;
239 enum {
240 size = 2,
241 alignment = Aligned16,
242 vectorizable = true,
243 masked_load_available = false,
244 masked_store_available = false
245 };
246 typedef Packet2d half;
247 typedef Packet2l integer_packet;
248};
249// The integer_packet of Packet2d, for the generic bit-manipulation paths; int64 is not vectorized.
250template <>
251struct unpacket_traits<Packet2l> {
252 using type = numext::int64_t;
253 using half = Packet2l;
254 static constexpr int size = 2;
255 static constexpr int alignment = Aligned16;
256 static constexpr bool vectorizable = false;
257 static constexpr bool masked_load_available = false;
258 static constexpr bool masked_store_available = false;
259};
260
261/* Forward declaration */
262EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock<Packet4f, 4>& kernel);
263
264inline std::ostream& operator<<(std::ostream& s, const Packet4i& v) {
265 Packet vt;
266 vt.v4i = v;
267 s << vt.i[0] << ", " << vt.i[1] << ", " << vt.i[2] << ", " << vt.i[3];
268 return s;
269}
270
271inline std::ostream& operator<<(std::ostream& s, const Packet4ui& v) {
272 Packet vt;
273 vt.v4ui = v;
274 s << vt.ui[0] << ", " << vt.ui[1] << ", " << vt.ui[2] << ", " << vt.ui[3];
275 return s;
276}
277
278inline std::ostream& operator<<(std::ostream& s, const Packet2l& v) {
279 Packet vt;
280 vt.v2l = v;
281 s << vt.l[0] << ", " << vt.l[1];
282 return s;
283}
284
285inline std::ostream& operator<<(std::ostream& s, const Packet2ul& v) {
286 Packet vt;
287 vt.v2ul = v;
288 s << vt.ul[0] << ", " << vt.ul[1];
289 return s;
290}
291
292inline std::ostream& operator<<(std::ostream& s, const Packet2d& v) {
293 Packet vt;
294 vt.v2d = v;
295 s << vt.d[0] << ", " << vt.d[1];
296 return s;
297}
298
299inline std::ostream& operator<<(std::ostream& s, const Packet4f& v) {
300 Packet vt;
301 vt.v4f = v;
302 s << vt.f[0] << ", " << vt.f[1] << ", " << vt.f[2] << ", " << vt.f[3];
303 return s;
304}
305
306template <>
307EIGEN_STRONG_INLINE Packet4i pload<Packet4i>(const int* from) {
308 EIGEN_DEBUG_ALIGNED_LOAD
309 return vec_xl(0, from);
310}
311
312template <>
313EIGEN_STRONG_INLINE Packet2d pload<Packet2d>(const double* from) {
314 EIGEN_DEBUG_ALIGNED_LOAD
315 return vec_xl(0, from);
316}
317
318template <>
319EIGEN_STRONG_INLINE void pstore<int>(int* to, const Packet4i& from) {
320 EIGEN_DEBUG_ALIGNED_STORE
321 vec_xst(from, 0, to);
322}
323
324template <>
325EIGEN_STRONG_INLINE void pstore<double>(double* to, const Packet2d& from) {
326 EIGEN_DEBUG_ALIGNED_STORE
327 vec_xst(from, 0, to);
328}
329
330template <>
331EIGEN_STRONG_INLINE Packet4f pfrexp<Packet4f>(const Packet4f& a, Packet4f& exponent) {
332 return pfrexp_generic(a, exponent);
333}
334
335template <>
336EIGEN_STRONG_INLINE Packet2d pfrexp<Packet2d>(const Packet2d& a, Packet2d& exponent) {
337 return pfrexp_generic(a, exponent);
338}
339
340template <>
341EIGEN_STRONG_INLINE Packet4i pset1<Packet4i>(const int& from) {
342 return vec_splats(from);
343}
344template <>
345EIGEN_STRONG_INLINE Packet2d pset1<Packet2d>(const double& from) {
346 return vec_splats(from);
347}
348template <>
349EIGEN_STRONG_INLINE Packet2l pset1<Packet2l>(const numext::int64_t& from) {
350 return vec_splats(static_cast<long long>(from));
351}
352
353template <>
354EIGEN_STRONG_INLINE void pbroadcast4<Packet4i>(const int* a, Packet4i& a0, Packet4i& a1, Packet4i& a2, Packet4i& a3) {
355 a3 = pload<Packet4i>(a);
356 a0 = vec_splat(a3, 0);
357 a1 = vec_splat(a3, 1);
358 a2 = vec_splat(a3, 2);
359 a3 = vec_splat(a3, 3);
360}
361
362template <>
363EIGEN_STRONG_INLINE void pbroadcast4<Packet2d>(const double* a, Packet2d& a0, Packet2d& a1, Packet2d& a2,
364 Packet2d& a3) {
365 a1 = pload<Packet2d>(a);
366 a0 = vec_splat(a1, 0);
367 a1 = vec_splat(a1, 1);
368 a3 = pload<Packet2d>(a + 2);
369 a2 = vec_splat(a3, 0);
370 a3 = vec_splat(a3, 1);
371}
372
373template <>
374EIGEN_DEVICE_FUNC inline Packet4i pgather<int, Packet4i>(const int* from, Index stride) {
375 EIGEN_ALIGN16 int ai[4];
376 ai[0] = from[0 * stride];
377 ai[1] = from[1 * stride];
378 ai[2] = from[2 * stride];
379 ai[3] = from[3 * stride];
380 return pload<Packet4i>(ai);
381}
382
383template <>
384EIGEN_DEVICE_FUNC inline Packet2d pgather<double, Packet2d>(const double* from, Index stride) {
385 EIGEN_ALIGN16 double af[2];
386 af[0] = from[0 * stride];
387 af[1] = from[1 * stride];
388 return pload<Packet2d>(af);
389}
390
391template <>
392EIGEN_DEVICE_FUNC inline void pscatter<int, Packet4i>(int* to, const Packet4i& from, Index stride) {
393 EIGEN_ALIGN16 int ai[4];
394 pstore<int>((int*)ai, from);
395 to[0 * stride] = ai[0];
396 to[1 * stride] = ai[1];
397 to[2 * stride] = ai[2];
398 to[3 * stride] = ai[3];
399}
400
401template <>
402EIGEN_DEVICE_FUNC inline void pscatter<double, Packet2d>(double* to, const Packet2d& from, Index stride) {
403 EIGEN_ALIGN16 double af[2];
404 pstore<double>(af, from);
405 to[0 * stride] = af[0];
406 to[1 * stride] = af[1];
407}
408
409template <>
410EIGEN_STRONG_INLINE Packet4i padd<Packet4i>(const Packet4i& a, const Packet4i& b) {
411 return (a + b);
412}
413template <>
414EIGEN_STRONG_INLINE Packet2d padd<Packet2d>(const Packet2d& a, const Packet2d& b) {
415 return (a + b);
416}
417
418template <>
419EIGEN_STRONG_INLINE Packet4i psub<Packet4i>(const Packet4i& a, const Packet4i& b) {
420 return (a - b);
421}
422template <>
423EIGEN_STRONG_INLINE Packet2d psub<Packet2d>(const Packet2d& a, const Packet2d& b) {
424 return (a - b);
425}
426
427template <>
428EIGEN_STRONG_INLINE Packet4i pmul<Packet4i>(const Packet4i& a, const Packet4i& b) {
429 return (a * b);
430}
431template <>
432EIGEN_STRONG_INLINE Packet2d pmul<Packet2d>(const Packet2d& a, const Packet2d& b) {
433 return (a * b);
434}
435
436template <>
437EIGEN_STRONG_INLINE Packet4i pdiv<Packet4i>(const Packet4i& a, const Packet4i& b) {
438 return (a / b);
439}
440template <>
441EIGEN_STRONG_INLINE Packet2d pdiv<Packet2d>(const Packet2d& a, const Packet2d& b) {
442 return (a / b);
443}
444
445template <>
446EIGEN_STRONG_INLINE Packet4i pnegate(const Packet4i& a) {
447 return (-a);
448}
449template <>
450EIGEN_STRONG_INLINE Packet2d pnegate(const Packet2d& a) {
451 return (-a);
452}
453
454template <>
455EIGEN_STRONG_INLINE Packet4i pmadd(const Packet4i& a, const Packet4i& b, const Packet4i& c) {
456 return padd<Packet4i>(pmul<Packet4i>(a, b), c);
457}
458template <>
459EIGEN_STRONG_INLINE Packet2d pmadd(const Packet2d& a, const Packet2d& b, const Packet2d& c) {
460 return vec_madd(a, b, c);
461}
462
463template <>
464EIGEN_STRONG_INLINE Packet4i plset<Packet4i>(const int& a) {
465 return padd<Packet4i>(pset1<Packet4i>(a), p4i_COUNTDOWN);
466}
467template <>
468EIGEN_STRONG_INLINE Packet2d plset<Packet2d>(const double& a) {
469 return padd<Packet2d>(pset1<Packet2d>(a), p2d_COUNTDOWN);
470}
471
472template <>
473EIGEN_STRONG_INLINE Packet4i pmin<Packet4i>(const Packet4i& a, const Packet4i& b) {
474 return vec_min(a, b);
475}
476// pmin/pmax follow std::min/std::max: they return a whenever either operand is NaN, so a NaN in a propagates through
477// clamps such as pmax(pmin(x, hi), lo). vec_min/vec_max (VFMIN/VFMAX mode 0 on z14) drop it.
478template <>
479EIGEN_STRONG_INLINE Packet2d pmin<Packet2d>(const Packet2d& a, const Packet2d& b) {
480 return __builtin_s390_vfmindb(a, b, 3); // mode 3: std::min
481}
482
483template <>
484EIGEN_STRONG_INLINE Packet4i pmax<Packet4i>(const Packet4i& a, const Packet4i& b) {
485 return vec_max(a, b);
486}
487template <>
488EIGEN_STRONG_INLINE Packet2d pmax<Packet2d>(const Packet2d& a, const Packet2d& b) {
489 return __builtin_s390_vfmaxdb(a, b, 3); // mode 3: std::max
490}
491
492template <>
493EIGEN_STRONG_INLINE Packet4i pand<Packet4i>(const Packet4i& a, const Packet4i& b) {
494 return vec_and(a, b);
495}
496template <>
497EIGEN_STRONG_INLINE Packet2d pand<Packet2d>(const Packet2d& a, const Packet2d& b) {
498 return vec_and(a, b);
499}
500
501template <>
502EIGEN_STRONG_INLINE Packet4i por<Packet4i>(const Packet4i& a, const Packet4i& b) {
503 return vec_or(a, b);
504}
505template <>
506EIGEN_STRONG_INLINE Packet2d por<Packet2d>(const Packet2d& a, const Packet2d& b) {
507 return vec_or(a, b);
508}
509
510template <>
511EIGEN_STRONG_INLINE Packet4i pxor<Packet4i>(const Packet4i& a, const Packet4i& b) {
512 return vec_xor(a, b);
513}
514template <>
515EIGEN_STRONG_INLINE Packet2d pxor<Packet2d>(const Packet2d& a, const Packet2d& b) {
516 return vec_xor(a, b);
517}
518
519template <>
520EIGEN_STRONG_INLINE Packet4i pandnot<Packet4i>(const Packet4i& a, const Packet4i& b) {
521 return pand<Packet4i>(a, vec_nor(b, b));
522}
523template <>
524EIGEN_STRONG_INLINE Packet2d pandnot<Packet2d>(const Packet2d& a, const Packet2d& b) {
525 return vec_and(a, vec_nor(b, b));
526}
527
528template <>
529EIGEN_STRONG_INLINE Packet2l padd<Packet2l>(const Packet2l& a, const Packet2l& b) {
530 return (a + b);
531}
532template <>
533EIGEN_STRONG_INLINE Packet2l psub<Packet2l>(const Packet2l& a, const Packet2l& b) {
534 return (a - b);
535}
536template <>
537EIGEN_STRONG_INLINE Packet2l pand<Packet2l>(const Packet2l& a, const Packet2l& b) {
538 return vec_and(a, b);
539}
540template <>
541EIGEN_STRONG_INLINE Packet2l por<Packet2l>(const Packet2l& a, const Packet2l& b) {
542 return vec_or(a, b);
543}
544template <>
545EIGEN_STRONG_INLINE Packet2l pxor<Packet2l>(const Packet2l& a, const Packet2l& b) {
546 return vec_xor(a, b);
547}
548template <>
549EIGEN_STRONG_INLINE Packet2l pandnot<Packet2l>(const Packet2l& a, const Packet2l& b) {
550 return vec_and(a, vec_nor(b, b));
551}
552template <>
553EIGEN_STRONG_INLINE Packet2l pcmp_eq<Packet2l>(const Packet2l& a, const Packet2l& b) {
554 return reinterpret_cast<Packet2l>(vec_cmpeq(a, b));
555}
556template <>
557EIGEN_STRONG_INLINE Packet2l pcmp_lt<Packet2l>(const Packet2l& a, const Packet2l& b) {
558 return reinterpret_cast<Packet2l>(vec_cmplt(a, b));
559}
560template <>
561EIGEN_STRONG_INLINE Packet2l pcmp_le<Packet2l>(const Packet2l& a, const Packet2l& b) {
562 return reinterpret_cast<Packet2l>(vec_cmple(a, b));
563}
564template <>
565EIGEN_STRONG_INLINE Packet2l pselect<Packet2l>(const Packet2l& mask, const Packet2l& a, const Packet2l& b) {
566 return vec_sel(b, a, reinterpret_cast<Packet2ul>(mask));
567}
568
569template <>
570struct cast_impl<Packet2l, Packet2d> {
571 EIGEN_DEVICE_FUNC static inline Packet2d run(const Packet2l& a) { return vec_double(a); }
572};
573
574template <>
575struct cast_impl<Packet2d, Packet2l> {
576 EIGEN_DEVICE_FUNC static inline Packet2l run(const Packet2d& a) {
577 return vec_signed(vec_sel(p2d_ZERO, a, vec_cmpeq(a, a))); // A NaN lane converts to 0.
578 }
579};
580
581template <>
582EIGEN_STRONG_INLINE Packet2d pround<Packet2d>(const Packet2d& a) {
583 /* Uses non-default rounding for vec_round */
584 return __builtin_s390_vfidb(a, 0, 1);
585}
586template <>
587EIGEN_STRONG_INLINE Packet2d pceil<Packet2d>(const Packet2d& a) {
588 return vec_ceil(a);
589}
590template <>
591EIGEN_STRONG_INLINE Packet2d pfloor<Packet2d>(const Packet2d& a) {
592 return vec_floor(a);
593}
594template <>
595EIGEN_STRONG_INLINE Packet2d print<Packet2d>(const Packet2d& a) {
596 // M5 = 4: round to nearest, ties to even.
597 return __builtin_s390_vfidb(a, 4, 4);
598}
599template <>
600EIGEN_STRONG_INLINE Packet2d ptrunc<Packet2d>(const Packet2d& a) {
601 // M5 = 5: round toward zero.
602 return __builtin_s390_vfidb(a, 4, 5);
603}
604template <>
605EIGEN_STRONG_INLINE Packet2d pcmp_lt_or_nan<Packet2d>(const Packet2d& a, const Packet2d& b) {
606 return pnot(pcmp_le(b, a));
607}
608
609template <>
610EIGEN_STRONG_INLINE Packet4i ploadu<Packet4i>(const int* from) {
611 return pload<Packet4i>(from);
612}
613template <>
614EIGEN_STRONG_INLINE Packet2d ploadu<Packet2d>(const double* from) {
615 return pload<Packet2d>(from);
616}
617
618template <>
619EIGEN_STRONG_INLINE Packet4i ploaddup<Packet4i>(const int* from) {
620 Packet4i p = pload<Packet4i>(from);
621 return vec_perm(p, p, p16uc_DUPLICATE32_HI);
622}
623
624template <>
625EIGEN_STRONG_INLINE Packet2d ploaddup<Packet2d>(const double* from) {
626 Packet2d p = pload<Packet2d>(from);
627 return vec_perm(p, p, p16uc_PSET64_HI);
628}
629
630template <>
631EIGEN_STRONG_INLINE void pstoreu<int>(int* to, const Packet4i& from) {
632 pstore<int>(to, from);
633}
634template <>
635EIGEN_STRONG_INLINE void pstoreu<double>(double* to, const Packet2d& from) {
636 pstore<double>(to, from);
637}
638
639template <>
640EIGEN_STRONG_INLINE void prefetch<int>(const int* addr) {
641 EIGEN_ZVECTOR_PREFETCH(addr);
642}
643template <>
644EIGEN_STRONG_INLINE void prefetch<double>(const double* addr) {
645 EIGEN_ZVECTOR_PREFETCH(addr);
646}
647
648template <int N>
649EIGEN_STRONG_INLINE Packet2l parithmetic_shift_right(const Packet2l& a) {
650 return Packet2l{parithmetic_shift_right<N>(a[0]), parithmetic_shift_right<N>(a[1])};
651}
652template <int N>
653EIGEN_STRONG_INLINE Packet4i parithmetic_shift_right(const Packet4i& a) {
654 return Packet4i{parithmetic_shift_right<N>(a[0]), parithmetic_shift_right<N>(a[1]), parithmetic_shift_right<N>(a[2]),
655 parithmetic_shift_right<N>(a[3])};
656}
657
658template <int N>
659EIGEN_STRONG_INLINE Packet2l plogical_shift_right(const Packet2l& a) {
660 return Packet2l{plogical_shift_right<N>(a[0]), plogical_shift_right<N>(a[1])};
661}
662template <int N>
663EIGEN_STRONG_INLINE Packet4i plogical_shift_right(const Packet4i& a) {
664 return Packet4i{plogical_shift_right<N>(a[0]), plogical_shift_right<N>(a[1]), plogical_shift_right<N>(a[2]),
665 plogical_shift_right<N>(a[3])};
666}
667
668template <int N>
669EIGEN_STRONG_INLINE Packet2l plogical_shift_left(const Packet2l& a) {
670 return Packet2l{plogical_shift_left<N>(a[0]), plogical_shift_left<N>(a[1])};
671}
672template <int N>
673EIGEN_STRONG_INLINE Packet4i plogical_shift_left(const Packet4i& a) {
674 return Packet4i{plogical_shift_left<N>(a[0]), plogical_shift_left<N>(a[1]), plogical_shift_left<N>(a[2]),
675 plogical_shift_left<N>(a[3])};
676}
677
678template <>
679EIGEN_STRONG_INLINE int pfirst<Packet4i>(const Packet4i& a) {
680 EIGEN_ALIGN16 int x[4];
681 pstore(x, a);
682 return x[0];
683}
684template <>
685EIGEN_STRONG_INLINE double pfirst<Packet2d>(const Packet2d& a) {
686 EIGEN_ALIGN16 double x[2];
687 pstore(x, a);
688 return x[0];
689}
690
691template <>
692EIGEN_STRONG_INLINE Packet4i preverse(const Packet4i& a) {
693 return reinterpret_cast<Packet4i>(
694 vec_perm(reinterpret_cast<Packet16uc>(a), reinterpret_cast<Packet16uc>(a), p16uc_REVERSE32));
695}
696
697template <>
698EIGEN_STRONG_INLINE Packet2d preverse(const Packet2d& a) {
699 return reinterpret_cast<Packet2d>(
700 vec_perm(reinterpret_cast<Packet16uc>(a), reinterpret_cast<Packet16uc>(a), p16uc_REVERSE64));
701}
702
703template <>
704EIGEN_STRONG_INLINE Packet4i pabs<Packet4i>(const Packet4i& a) {
705 return vec_abs(a);
706}
707template <>
708EIGEN_STRONG_INLINE Packet2d pabs<Packet2d>(const Packet2d& a) {
709 return vec_abs(a);
710}
711
712template <>
713EIGEN_STRONG_INLINE int predux<Packet4i>(const Packet4i& a) {
714 Packet4i b, sum;
715 b = vec_sld(a, a, 8);
716 sum = padd<Packet4i>(a, b);
717 b = vec_sld(sum, sum, 4);
718 sum = padd<Packet4i>(sum, b);
719 return pfirst(sum);
720}
721
722template <>
723EIGEN_STRONG_INLINE double predux<Packet2d>(const Packet2d& a) {
724 Packet2d b, sum;
725 b = reinterpret_cast<Packet2d>(vec_sld(reinterpret_cast<Packet4i>(a), reinterpret_cast<Packet4i>(a), 8));
726 sum = padd<Packet2d>(a, b);
727 return pfirst(sum);
728}
729
730template <>
731EIGEN_STRONG_INLINE bool predux_any(const Packet2d& a) {
732 const Packet2ul zero = {0, 0};
733 return vec_any_ne(reinterpret_cast<Packet2ul>(a), zero);
734}
735
736// Other reduction functions:
737// mul
738template <>
739EIGEN_STRONG_INLINE int predux_mul<Packet4i>(const Packet4i& a) {
740 EIGEN_ALIGN16 int aux[4];
741 pstore(aux, a);
742 return aux[0] * aux[1] * aux[2] * aux[3];
743}
744
745template <>
746EIGEN_STRONG_INLINE double predux_mul<Packet2d>(const Packet2d& a) {
747 return pfirst(
748 pmul(a, reinterpret_cast<Packet2d>(vec_sld(reinterpret_cast<Packet4i>(a), reinterpret_cast<Packet4i>(a), 8))));
749}
750
751// min
752template <>
753EIGEN_STRONG_INLINE int predux_min<Packet4i>(const Packet4i& a) {
754 Packet4i b, res;
755 b = pmin<Packet4i>(a, vec_sld(a, a, 8));
756 res = pmin<Packet4i>(b, vec_sld(b, b, 4));
757 return pfirst(res);
758}
759
760template <>
761EIGEN_STRONG_INLINE double predux_min<Packet2d>(const Packet2d& a) {
762 return pfirst(pmin<Packet2d>(
763 a, reinterpret_cast<Packet2d>(vec_sld(reinterpret_cast<Packet4i>(a), reinterpret_cast<Packet4i>(a), 8))));
764}
765
766// max
767template <>
768EIGEN_STRONG_INLINE int predux_max<Packet4i>(const Packet4i& a) {
769 Packet4i b, res;
770 b = pmax<Packet4i>(a, vec_sld(a, a, 8));
771 res = pmax<Packet4i>(b, vec_sld(b, b, 4));
772 return pfirst(res);
773}
774
775// max
776template <>
777EIGEN_STRONG_INLINE double predux_max<Packet2d>(const Packet2d& a) {
778 return pfirst(pmax<Packet2d>(
779 a, reinterpret_cast<Packet2d>(vec_sld(reinterpret_cast<Packet4i>(a), reinterpret_cast<Packet4i>(a), 8))));
780}
781
782EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock<Packet4i, 4>& kernel) {
783 Packet4i t0 = vec_mergeh(kernel.packet[0], kernel.packet[2]);
784 Packet4i t1 = vec_mergel(kernel.packet[0], kernel.packet[2]);
785 Packet4i t2 = vec_mergeh(kernel.packet[1], kernel.packet[3]);
786 Packet4i t3 = vec_mergel(kernel.packet[1], kernel.packet[3]);
787 kernel.packet[0] = vec_mergeh(t0, t2);
788 kernel.packet[1] = vec_mergel(t0, t2);
789 kernel.packet[2] = vec_mergeh(t1, t3);
790 kernel.packet[3] = vec_mergel(t1, t3);
791}
792
793EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock<Packet2d, 2>& kernel) {
794 Packet2d t0 = vec_perm(kernel.packet[0], kernel.packet[1], p16uc_TRANSPOSE64_HI);
795 Packet2d t1 = vec_perm(kernel.packet[0], kernel.packet[1], p16uc_TRANSPOSE64_LO);
796 kernel.packet[0] = t0;
797 kernel.packet[1] = t1;
798}
799
800template <>
801EIGEN_STRONG_INLINE Packet4f pload<Packet4f>(const float* from) {
802 EIGEN_DEBUG_ALIGNED_LOAD
803 return vec_xl(0, from);
804}
805
806template <>
807EIGEN_STRONG_INLINE void pstore<float>(float* to, const Packet4f& from) {
808 EIGEN_DEBUG_ALIGNED_STORE
809 vec_xst(from, 0, to);
810}
811
812template <>
813EIGEN_STRONG_INLINE Packet4f pset1<Packet4f>(const float& from) {
814 return vec_splats(from);
815}
816
817template <>
818EIGEN_STRONG_INLINE void pbroadcast4<Packet4f>(const float* a, Packet4f& a0, Packet4f& a1, Packet4f& a2, Packet4f& a3) {
819 a3 = pload<Packet4f>(a);
820 a0 = vec_splat(a3, 0);
821 a1 = vec_splat(a3, 1);
822 a2 = vec_splat(a3, 2);
823 a3 = vec_splat(a3, 3);
824}
825
826template <>
827EIGEN_DEVICE_FUNC inline Packet4f pgather<float, Packet4f>(const float* from, Index stride) {
828 EIGEN_ALIGN16 float af[4];
829 af[0] = from[0 * stride];
830 af[1] = from[1 * stride];
831 af[2] = from[2 * stride];
832 af[3] = from[3 * stride];
833 return pload<Packet4f>(af);
834}
835
836template <>
837EIGEN_DEVICE_FUNC inline void pscatter<float, Packet4f>(float* to, const Packet4f& from, Index stride) {
838 EIGEN_ALIGN16 float af[4];
839 pstore<float>((float*)af, from);
840 to[0 * stride] = af[0];
841 to[1 * stride] = af[1];
842 to[2 * stride] = af[2];
843 to[3 * stride] = af[3];
844}
845
846template <>
847EIGEN_STRONG_INLINE Packet4f padd<Packet4f>(const Packet4f& a, const Packet4f& b) {
848 return (a + b);
849}
850template <>
851EIGEN_STRONG_INLINE Packet4f psub<Packet4f>(const Packet4f& a, const Packet4f& b) {
852 return (a - b);
853}
854template <>
855EIGEN_STRONG_INLINE Packet4f pmul<Packet4f>(const Packet4f& a, const Packet4f& b) {
856 return (a * b);
857}
858template <>
859EIGEN_STRONG_INLINE Packet4f pdiv<Packet4f>(const Packet4f& a, const Packet4f& b) {
860 return (a / b);
861}
862template <>
863EIGEN_STRONG_INLINE Packet4f pnegate<Packet4f>(const Packet4f& a) {
864 return (-a);
865}
866template <>
867EIGEN_STRONG_INLINE Packet4f pmadd<Packet4f>(const Packet4f& a, const Packet4f& b, const Packet4f& c) {
868 return vec_madd(a, b, c);
869}
870template <>
871EIGEN_STRONG_INLINE Packet4f pmin<Packet4f>(const Packet4f& a, const Packet4f& b) {
872 return __builtin_s390_vfminsb(a, b, 3); // mode 3: std::min
873}
874template <>
875EIGEN_STRONG_INLINE Packet4f pmax<Packet4f>(const Packet4f& a, const Packet4f& b) {
876 return __builtin_s390_vfmaxsb(a, b, 3); // mode 3: std::max
877}
878template <>
879EIGEN_STRONG_INLINE Packet4f pand<Packet4f>(const Packet4f& a, const Packet4f& b) {
880 return vec_and(a, b);
881}
882template <>
883EIGEN_STRONG_INLINE Packet4f por<Packet4f>(const Packet4f& a, const Packet4f& b) {
884 return vec_or(a, b);
885}
886template <>
887EIGEN_STRONG_INLINE Packet4f pxor<Packet4f>(const Packet4f& a, const Packet4f& b) {
888 return vec_xor(a, b);
889}
890template <>
891EIGEN_STRONG_INLINE Packet4f pandnot<Packet4f>(const Packet4f& a, const Packet4f& b) {
892 return vec_and(a, vec_nor(b, b));
893}
894template <>
895EIGEN_STRONG_INLINE Packet4f pround<Packet4f>(const Packet4f& a) {
896 /* Uses non-default rounding for vec_round */
897 return __builtin_s390_vfisb(a, 0, 1);
898}
899template <>
900EIGEN_STRONG_INLINE Packet4f pceil<Packet4f>(const Packet4f& a) {
901 return vec_ceil(a);
902}
903template <>
904EIGEN_STRONG_INLINE Packet4f pfloor<Packet4f>(const Packet4f& a) {
905 return vec_floor(a);
906}
907template <>
908EIGEN_STRONG_INLINE Packet4f print<Packet4f>(const Packet4f& a) {
909 // M5 = 4: round to nearest, ties to even.
910 return __builtin_s390_vfisb(a, 4, 4);
911}
912template <>
913EIGEN_STRONG_INLINE Packet4f ptrunc<Packet4f>(const Packet4f& a) {
914 // M5 = 5: round toward zero.
915 return __builtin_s390_vfisb(a, 4, 5);
916}
917template <>
918EIGEN_STRONG_INLINE Packet4f pcmp_lt_or_nan<Packet4f>(const Packet4f& a, const Packet4f& b) {
919 return pnot(pcmp_le(b, a));
920}
921template <>
922EIGEN_STRONG_INLINE Packet4f pabs<Packet4f>(const Packet4f& a) {
923 return vec_abs(a);
924}
925template <>
926EIGEN_STRONG_INLINE float pfirst<Packet4f>(const Packet4f& a) {
927 EIGEN_ALIGN16 float x[4];
928 pstore(x, a);
929 return x[0];
930}
931
932template <>
933EIGEN_STRONG_INLINE Packet4f ploaddup<Packet4f>(const float* from) {
934 Packet4f p = pload<Packet4f>(from);
935 return vec_perm(p, p, p16uc_DUPLICATE32_HI);
936}
937
938template <>
939EIGEN_STRONG_INLINE Packet4f preverse(const Packet4f& a) {
940 return reinterpret_cast<Packet4f>(
941 vec_perm(reinterpret_cast<Packet16uc>(a), reinterpret_cast<Packet16uc>(a), p16uc_REVERSE32));
942}
943
944template <>
945EIGEN_STRONG_INLINE float predux<Packet4f>(const Packet4f& a) {
946 Packet4f b, sum;
947 b = vec_sld(a, a, 8);
948 sum = padd<Packet4f>(a, b);
949 b = vec_sld(sum, sum, 4);
950 sum = padd<Packet4f>(sum, b);
951 return pfirst(sum);
952}
953
954template <>
955EIGEN_STRONG_INLINE bool predux_any(const Packet4f& a) {
956 const Packet4ui zero = {0, 0, 0, 0};
957 return vec_any_ne(reinterpret_cast<Packet4ui>(a), zero);
958}
959
960// Other reduction functions:
961// mul
962template <>
963EIGEN_STRONG_INLINE float predux_mul<Packet4f>(const Packet4f& a) {
964 Packet4f prod;
965 prod = pmul(a, vec_sld(a, a, 8));
966 return pfirst(pmul(prod, vec_sld(prod, prod, 4)));
967}
968
969// min
970template <>
971EIGEN_STRONG_INLINE float predux_min<Packet4f>(const Packet4f& a) {
972 Packet4f b, res;
973 b = pmin<Packet4f>(a, vec_sld(a, a, 8));
974 res = pmin<Packet4f>(b, vec_sld(b, b, 4));
975 return pfirst(res);
976}
977
978// max
979template <>
980EIGEN_STRONG_INLINE float predux_max<Packet4f>(const Packet4f& a) {
981 Packet4f b, res;
982 b = pmax<Packet4f>(a, vec_sld(a, a, 8));
983 res = pmax<Packet4f>(b, vec_sld(b, b, 4));
984 return pfirst(res);
985}
986
987EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock<Packet4f, 4>& kernel) {
988 Packet4f t0 = vec_mergeh(kernel.packet[0], kernel.packet[2]);
989 Packet4f t1 = vec_mergel(kernel.packet[0], kernel.packet[2]);
990 Packet4f t2 = vec_mergeh(kernel.packet[1], kernel.packet[3]);
991 Packet4f t3 = vec_mergel(kernel.packet[1], kernel.packet[3]);
992 kernel.packet[0] = vec_mergeh(t0, t2);
993 kernel.packet[1] = vec_mergel(t0, t2);
994 kernel.packet[2] = vec_mergeh(t1, t3);
995 kernel.packet[3] = vec_mergel(t1, t3);
996}
997
998template <>
999EIGEN_STRONG_INLINE Packet4f pldexp<Packet4f>(const Packet4f& a, const Packet4f& exponent) {
1000 return pldexp_generic(a, exponent);
1001}
1002
1003template <>
1004EIGEN_STRONG_INLINE Packet2d pldexp<Packet2d>(const Packet2d& a, const Packet2d& exponent) {
1005 // The single-rounding split of pldexp_generic on int64 exponents.
1006 const Packet2d max_exponent = pset1<Packet2d>(2099.0);
1007 const Packet2d last_max = pset1<Packet2d>(1022.0);
1008 const Packet2l e = pcast<Packet2d, Packet2l>(pmin(pmax(exponent, pnegate(max_exponent)), max_exponent));
1009 const Packet2l b = pcast<Packet2d, Packet2l>(pmin(pmax(exponent, pnegate(last_max)), last_max));
1010 const Packet2l bias = {1023, 1023};
1011 const Packet2l even = {-2, -2};
1012 const Packet2l t = psub(e, b) & even;
1013 const Packet2d c1 = reinterpret_cast<Packet2d>(plogical_shift_left<51>(t + bias + bias)); // 2^(t/2)
1014 const Packet2d c2 = reinterpret_cast<Packet2d>(plogical_shift_left<52>(psub(e, t) + bias)); // 2^(e - t)
1015 return pldexp_apply_factors(a, c1, c2); // a * 2^e
1016}
1017
1018template <>
1019EIGEN_STRONG_INLINE void prefetch<float>(const float* addr) {
1020 EIGEN_ZVECTOR_PREFETCH(addr);
1021}
1022template <>
1023EIGEN_STRONG_INLINE Packet4f ploadu<Packet4f>(const float* from) {
1024 return pload<Packet4f>(from);
1025}
1026template <>
1027EIGEN_STRONG_INLINE void pstoreu<float>(float* to, const Packet4f& from) {
1028 pstore<float>(to, from);
1029}
1030template <>
1031EIGEN_STRONG_INLINE Packet4f plset<Packet4f>(const float& a) {
1032 return padd<Packet4f>(pset1<Packet4f>(a), p4f_COUNTDOWN);
1033}
1034
1035#if !defined(vec_float) || !defined(__ARCH__) || (defined(__ARCH__) && __ARCH__ < 13)
1036#pragma GCC warning "float->int and int->float conversion is simulated. compile for z15 for improved performance"
1037template <>
1038struct cast_impl<Packet4i, Packet4f> {
1039 EIGEN_DEVICE_FUNC static inline Packet4f run(const Packet4i& a) {
1040 return Packet4f{float(a[0]), float(a[1]), float(a[2]), float(a[3])};
1041 }
1042};
1043
1044template <>
1045struct cast_impl<Packet4f, Packet4i> {
1046 // A NaN lane converts to 0: the scalar conversion would be undefined, and pexp's clamp lets NaN through.
1047 EIGEN_DEVICE_FUNC static inline Packet4i run(const Packet4f& a) {
1048 return Packet4i{to_int(a[0]), to_int(a[1]), to_int(a[2]), to_int(a[3])};
1049 }
1050 EIGEN_DEVICE_FUNC static inline int to_int(float x) { return x == x ? int(x) : 0; }
1051};
1052#else
1053template <>
1054struct cast_impl<Packet4i, Packet4f> {
1055 EIGEN_DEVICE_FUNC static inline Packet4f run(const Packet4i& a) { return vec_float(a); }
1056};
1057
1058template <>
1059struct cast_impl<Packet4f, Packet4i> {
1060 EIGEN_DEVICE_FUNC static inline Packet4i run(const Packet4f& a) {
1061 return vec_signed(vec_sel(p4f_ZERO, a, vec_cmpeq(a, a))); // A NaN lane converts to 0.
1062 }
1063};
1064#endif
1065
1066template <>
1067EIGEN_STRONG_INLINE Packet4f pset1frombits<Packet4f>(uint32_t from) {
1068 return pset1<Packet4f>(Eigen::numext::bit_cast<float>(from));
1069}
1070template <>
1071EIGEN_STRONG_INLINE Packet2d pset1frombits<Packet2d>(uint64_t from) {
1072 return pset1<Packet2d>(Eigen::numext::bit_cast<double>(from));
1073}
1074
1075} // end namespace internal
1076
1077} // end namespace Eigen
1078
1079#endif // EIGEN_PACKET_MATH_ZVECTOR_H
@ Aligned16
Definition Constants.h:238