Eigen  5.0.1
 
Loading...
Searching...
No Matches
PacketMath.h
1// This file is part of Eigen, a lightweight C++ template library
2// for linear algebra.
3//
4// Copyright (C) 2018 Wave Computing, Inc.
5// Written by:
6// Chris Larsen
7// Alexey Frunze (afrunze@wavecomp.com)
8//
9// This Source Code Form is subject to the terms of the Mozilla
10// Public License v. 2.0. If a copy of the MPL was not distributed
11// with this file, You can obtain one at http://mozilla.org/MPL/2.0/.
12// SPDX-License-Identifier: MPL-2.0
13
14#ifndef EIGEN_PACKET_MATH_MSA_H
15#define EIGEN_PACKET_MATH_MSA_H
16
17#include <iostream>
18#include <string>
19
20// IWYU pragma: private
21#include "../../InternalHeaderCheck.h"
22
23namespace Eigen {
24
25namespace internal {
26
27#ifndef EIGEN_CACHEFRIENDLY_PRODUCT_THRESHOLD
28#define EIGEN_CACHEFRIENDLY_PRODUCT_THRESHOLD 8
29#endif
30
31#ifndef EIGEN_HAS_SINGLE_INSTRUCTION_MADD
32#define EIGEN_HAS_SINGLE_INSTRUCTION_MADD
33#endif
34
35#ifndef EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS
36#define EIGEN_ARCH_DEFAULT_NUMBER_OF_REGISTERS 32
37#endif
38
39#define EIGEN_MSA_DEBUG
40
41#define EIGEN_MSA_SHF_I8(a, b, c, d) (((d) << 6) | ((c) << 4) | ((b) << 2) | (a))
42
43typedef v4f32 Packet4f;
44typedef v4i32 Packet4i;
45typedef v4u32 Packet4ui;
46
47#define EIGEN_DECLARE_CONST_Packet4f(NAME, X) const Packet4f p4f_##NAME = {X, X, X, X}
48#define EIGEN_DECLARE_CONST_Packet4i(NAME, X) const Packet4i p4i_##NAME = {X, X, X, X}
49#define EIGEN_DECLARE_CONST_Packet4ui(NAME, X) const Packet4ui p4ui_##NAME = {X, X, X, X}
50
51inline std::ostream& operator<<(std::ostream& os, const Packet4f& value) {
52 os << "[ " << value[0] << ", " << value[1] << ", " << value[2] << ", " << value[3] << " ]";
53 return os;
54}
55
56inline std::ostream& operator<<(std::ostream& os, const Packet4i& value) {
57 os << "[ " << value[0] << ", " << value[1] << ", " << value[2] << ", " << value[3] << " ]";
58 return os;
59}
60
61inline std::ostream& operator<<(std::ostream& os, const Packet4ui& value) {
62 os << "[ " << value[0] << ", " << value[1] << ", " << value[2] << ", " << value[3] << " ]";
63 return os;
64}
65
66template <>
67struct packet_traits<float> : default_packet_traits {
68 typedef Packet4f type;
69 typedef Packet4f half; // Packet2f intrinsics not implemented yet
70 enum {
71 Vectorizable = 1,
72 AlignedOnScalar = 1,
73 size = 4,
74 // FIXME: verify the Has* flags.
75 HasDiv = 1,
76 HasSin = EIGEN_FAST_MATH,
77 HasCos = EIGEN_FAST_MATH,
78 HasTanh = EIGEN_FAST_MATH,
79 HasErf = EIGEN_FAST_MATH,
80 HasLog = 1,
81 HasExp = 1,
82 HasSqrt = 1,
83 HasRsqrt = 1,
84 };
85};
86
87template <>
88struct packet_traits<int32_t> : default_packet_traits {
89 typedef Packet4i type;
90 typedef Packet4i half; // Packet2i intrinsics not implemented yet
91 enum {
92 Vectorizable = 1,
93 AlignedOnScalar = 1,
94 size = 4,
95 // FIXME: verify the Has* flags.
96 HasDiv = 1,
97 };
98};
99
100template <>
101struct unpacket_traits<Packet4f> {
102 typedef float type;
103 enum {
104 size = 4,
105 alignment = Aligned16,
106 vectorizable = true,
107 masked_load_available = false,
108 masked_store_available = false
109 };
110 typedef Packet4f half;
111};
112
113template <>
114struct unpacket_traits<Packet4i> {
115 typedef int32_t type;
116 enum {
117 size = 4,
118 alignment = Aligned16,
119 vectorizable = true,
120 masked_load_available = false,
121 masked_store_available = false
122 };
123 typedef Packet4i half;
124};
125
126template <>
127EIGEN_STRONG_INLINE Packet4f pset1<Packet4f>(const float& from) {
128 EIGEN_MSA_DEBUG;
129
130 Packet4f v = {from, from, from, from};
131 return v;
132}
133
134template <>
135EIGEN_STRONG_INLINE Packet4i pset1<Packet4i>(const int32_t& from) {
136 EIGEN_MSA_DEBUG;
137
138 return __builtin_msa_fill_w(from);
139}
140
141template <>
142EIGEN_STRONG_INLINE Packet4f pload1<Packet4f>(const float* from) {
143 EIGEN_MSA_DEBUG;
144
145 float f = *from;
146 Packet4f v = {f, f, f, f};
147 return v;
148}
149
150template <>
151EIGEN_STRONG_INLINE Packet4i pload1<Packet4i>(const int32_t* from) {
152 EIGEN_MSA_DEBUG;
153
154 return __builtin_msa_fill_w(*from);
155}
156
157template <>
158EIGEN_STRONG_INLINE Packet4f padd<Packet4f>(const Packet4f& a, const Packet4f& b) {
159 EIGEN_MSA_DEBUG;
160
161 return __builtin_msa_fadd_w(a, b);
162}
163
164template <>
165EIGEN_STRONG_INLINE Packet4i padd<Packet4i>(const Packet4i& a, const Packet4i& b) {
166 EIGEN_MSA_DEBUG;
167
168 return __builtin_msa_addv_w(a, b);
169}
170
171template <>
172EIGEN_STRONG_INLINE Packet4f plset<Packet4f>(const float& a) {
173 EIGEN_MSA_DEBUG;
174
175 static const Packet4f countdown = {0.0f, 1.0f, 2.0f, 3.0f};
176 return padd(pset1<Packet4f>(a), countdown);
177}
178
179template <>
180EIGEN_STRONG_INLINE Packet4i plset<Packet4i>(const int32_t& a) {
181 EIGEN_MSA_DEBUG;
182
183 static const Packet4i countdown = {0, 1, 2, 3};
184 return padd(pset1<Packet4i>(a), countdown);
185}
186
187template <>
188EIGEN_STRONG_INLINE Packet4f psub<Packet4f>(const Packet4f& a, const Packet4f& b) {
189 EIGEN_MSA_DEBUG;
190
191 return __builtin_msa_fsub_w(a, b);
192}
193
194template <>
195EIGEN_STRONG_INLINE Packet4i psub<Packet4i>(const Packet4i& a, const Packet4i& b) {
196 EIGEN_MSA_DEBUG;
197
198 return __builtin_msa_subv_w(a, b);
199}
200
201template <>
202EIGEN_STRONG_INLINE Packet4f pnegate(const Packet4f& a) {
203 EIGEN_MSA_DEBUG;
204
205 return (Packet4f)__builtin_msa_bnegi_w((v4u32)a, 31);
206}
207
208template <>
209EIGEN_STRONG_INLINE Packet4i pnegate(const Packet4i& a) {
210 EIGEN_MSA_DEBUG;
211
212 return __builtin_msa_addvi_w((v4i32)__builtin_msa_nori_b((v16u8)a, 0), 1);
213}
214
215template <>
216EIGEN_STRONG_INLINE Packet4f pmul<Packet4f>(const Packet4f& a, const Packet4f& b) {
217 EIGEN_MSA_DEBUG;
218
219 return __builtin_msa_fmul_w(a, b);
220}
221
222template <>
223EIGEN_STRONG_INLINE Packet4i pmul<Packet4i>(const Packet4i& a, const Packet4i& b) {
224 EIGEN_MSA_DEBUG;
225
226 return __builtin_msa_mulv_w(a, b);
227}
228
229template <>
230EIGEN_STRONG_INLINE Packet4f pdiv<Packet4f>(const Packet4f& a, const Packet4f& b) {
231 EIGEN_MSA_DEBUG;
232
233 return __builtin_msa_fdiv_w(a, b);
234}
235
236template <>
237EIGEN_STRONG_INLINE Packet4i pdiv<Packet4i>(const Packet4i& a, const Packet4i& b) {
238 EIGEN_MSA_DEBUG;
239
240 return __builtin_msa_div_s_w(a, b);
241}
242
243template <>
244EIGEN_STRONG_INLINE Packet4f pmadd(const Packet4f& a, const Packet4f& b, const Packet4f& c) {
245 EIGEN_MSA_DEBUG;
246
247 return __builtin_msa_fmadd_w(c, a, b);
248}
249
250template <>
251EIGEN_STRONG_INLINE Packet4i pmadd(const Packet4i& a, const Packet4i& b, const Packet4i& c) {
252 EIGEN_MSA_DEBUG;
253
254 // Use "asm" construct to avoid __builtin_msa_maddv_w GNU C bug.
255 Packet4i value = c;
256 __asm__("maddv.w %w[value], %w[a], %w[b]\n"
257 // Outputs
258 : [value] "+f"(value)
259 // Inputs
260 : [a] "f"(a), [b] "f"(b));
261 return value;
262}
263
264template <>
265EIGEN_STRONG_INLINE Packet4f pand<Packet4f>(const Packet4f& a, const Packet4f& b) {
266 EIGEN_MSA_DEBUG;
267
268 return (Packet4f)__builtin_msa_and_v((v16u8)a, (v16u8)b);
269}
270
271template <>
272EIGEN_STRONG_INLINE Packet4i pand<Packet4i>(const Packet4i& a, const Packet4i& b) {
273 EIGEN_MSA_DEBUG;
274
275 return (Packet4i)__builtin_msa_and_v((v16u8)a, (v16u8)b);
276}
277
278template <>
279EIGEN_STRONG_INLINE Packet4f por<Packet4f>(const Packet4f& a, const Packet4f& b) {
280 EIGEN_MSA_DEBUG;
281
282 return (Packet4f)__builtin_msa_or_v((v16u8)a, (v16u8)b);
283}
284
285template <>
286EIGEN_STRONG_INLINE Packet4i por<Packet4i>(const Packet4i& a, const Packet4i& b) {
287 EIGEN_MSA_DEBUG;
288
289 return (Packet4i)__builtin_msa_or_v((v16u8)a, (v16u8)b);
290}
291
292template <>
293EIGEN_STRONG_INLINE Packet4f pxor<Packet4f>(const Packet4f& a, const Packet4f& b) {
294 EIGEN_MSA_DEBUG;
295
296 return (Packet4f)__builtin_msa_xor_v((v16u8)a, (v16u8)b);
297}
298
299template <>
300EIGEN_STRONG_INLINE Packet4i pxor<Packet4i>(const Packet4i& a, const Packet4i& b) {
301 EIGEN_MSA_DEBUG;
302
303 return (Packet4i)__builtin_msa_xor_v((v16u8)a, (v16u8)b);
304}
305
306template <>
307EIGEN_STRONG_INLINE Packet4f pandnot<Packet4f>(const Packet4f& a, const Packet4f& b) {
308 EIGEN_MSA_DEBUG;
309
310 return pand(a, (Packet4f)__builtin_msa_xori_b((v16u8)b, 255));
311}
312
313template <>
314EIGEN_STRONG_INLINE Packet4i pandnot<Packet4i>(const Packet4i& a, const Packet4i& b) {
315 EIGEN_MSA_DEBUG;
316
317 return pand(a, (Packet4i)__builtin_msa_xori_b((v16u8)b, 255));
318}
319
320template <>
321EIGEN_STRONG_INLINE Packet4f pmin<Packet4f>(const Packet4f& a, const Packet4f& b) {
322 EIGEN_MSA_DEBUG;
323
324#if EIGEN_FAST_MATH
325 // This prefers numbers to NaNs.
326 return __builtin_msa_fmin_w(a, b);
327#else
328 // This prefers NaNs to numbers.
329 Packet4i aNaN = __builtin_msa_fcun_w(a, a);
330 Packet4i aMinOrNaN = por(__builtin_msa_fclt_w(a, b), aNaN);
331 return (Packet4f)__builtin_msa_bsel_v((v16u8)aMinOrNaN, (v16u8)b, (v16u8)a);
332#endif
333}
334
335template <>
336EIGEN_STRONG_INLINE Packet4i pmin<Packet4i>(const Packet4i& a, const Packet4i& b) {
337 EIGEN_MSA_DEBUG;
338
339 return __builtin_msa_min_s_w(a, b);
340}
341
342template <>
343EIGEN_STRONG_INLINE Packet4f pmax<Packet4f>(const Packet4f& a, const Packet4f& b) {
344 EIGEN_MSA_DEBUG;
345
346#if EIGEN_FAST_MATH
347 // This prefers numbers to NaNs.
348 return __builtin_msa_fmax_w(a, b);
349#else
350 // This prefers NaNs to numbers.
351 Packet4i aNaN = __builtin_msa_fcun_w(a, a);
352 Packet4i aMaxOrNaN = por(__builtin_msa_fclt_w(b, a), aNaN);
353 return (Packet4f)__builtin_msa_bsel_v((v16u8)aMaxOrNaN, (v16u8)b, (v16u8)a);
354#endif
355}
356
357template <>
358EIGEN_STRONG_INLINE Packet4i pmax<Packet4i>(const Packet4i& a, const Packet4i& b) {
359 EIGEN_MSA_DEBUG;
360
361 return __builtin_msa_max_s_w(a, b);
362}
363
364template <>
365EIGEN_STRONG_INLINE Packet4f pload<Packet4f>(const float* from) {
366 EIGEN_MSA_DEBUG;
367
368 EIGEN_DEBUG_ALIGNED_LOAD return (Packet4f)__builtin_msa_ld_w(const_cast<float*>(from), 0);
369}
370
371template <>
372EIGEN_STRONG_INLINE Packet4i pload<Packet4i>(const int32_t* from) {
373 EIGEN_MSA_DEBUG;
374
375 EIGEN_DEBUG_ALIGNED_LOAD return __builtin_msa_ld_w(const_cast<int32_t*>(from), 0);
376}
377
378template <>
379EIGEN_STRONG_INLINE Packet4f ploadu<Packet4f>(const float* from) {
380 EIGEN_MSA_DEBUG;
381
382 EIGEN_DEBUG_UNALIGNED_LOAD return (Packet4f)__builtin_msa_ld_w(const_cast<float*>(from), 0);
383}
384
385template <>
386EIGEN_STRONG_INLINE Packet4i ploadu<Packet4i>(const int32_t* from) {
387 EIGEN_MSA_DEBUG;
388
389 EIGEN_DEBUG_UNALIGNED_LOAD return (Packet4i)__builtin_msa_ld_w(const_cast<int32_t*>(from), 0);
390}
391
392template <>
393EIGEN_STRONG_INLINE Packet4f ploaddup<Packet4f>(const float* from) {
394 EIGEN_MSA_DEBUG;
395
396 float f0 = from[0], f1 = from[1];
397 Packet4f v0 = {f0, f0, f0, f0};
398 Packet4f v1 = {f1, f1, f1, f1};
399 return (Packet4f)__builtin_msa_ilvr_d((v2i64)v1, (v2i64)v0);
400}
401
402template <>
403EIGEN_STRONG_INLINE Packet4i ploaddup<Packet4i>(const int32_t* from) {
404 EIGEN_MSA_DEBUG;
405
406 int32_t i0 = from[0], i1 = from[1];
407 Packet4i v0 = {i0, i0, i0, i0};
408 Packet4i v1 = {i1, i1, i1, i1};
409 return (Packet4i)__builtin_msa_ilvr_d((v2i64)v1, (v2i64)v0);
410}
411
412template <>
413EIGEN_STRONG_INLINE void pstore<float>(float* to, const Packet4f& from) {
414 EIGEN_MSA_DEBUG;
415
416 EIGEN_DEBUG_ALIGNED_STORE __builtin_msa_st_w((Packet4i)from, to, 0);
417}
418
419template <>
420EIGEN_STRONG_INLINE void pstore<int32_t>(int32_t* to, const Packet4i& from) {
421 EIGEN_MSA_DEBUG;
422
423 EIGEN_DEBUG_ALIGNED_STORE __builtin_msa_st_w(from, to, 0);
424}
425
426template <>
427EIGEN_STRONG_INLINE void pstoreu<float>(float* to, const Packet4f& from) {
428 EIGEN_MSA_DEBUG;
429
430 EIGEN_DEBUG_UNALIGNED_STORE __builtin_msa_st_w((Packet4i)from, to, 0);
431}
432
433template <>
434EIGEN_STRONG_INLINE void pstoreu<int32_t>(int32_t* to, const Packet4i& from) {
435 EIGEN_MSA_DEBUG;
436
437 EIGEN_DEBUG_UNALIGNED_STORE __builtin_msa_st_w(from, to, 0);
438}
439
440template <>
441EIGEN_DEVICE_FUNC inline Packet4f pgather<float, Packet4f>(const float* from, Index stride) {
442 EIGEN_MSA_DEBUG;
443
444 float f = *from;
445 Packet4f v = {f, f, f, f};
446 v[1] = from[stride];
447 v[2] = from[2 * stride];
448 v[3] = from[3 * stride];
449 return v;
450}
451
452template <>
453EIGEN_DEVICE_FUNC inline Packet4i pgather<int32_t, Packet4i>(const int32_t* from, Index stride) {
454 EIGEN_MSA_DEBUG;
455
456 int32_t i = *from;
457 Packet4i v = {i, i, i, i};
458 v[1] = from[stride];
459 v[2] = from[2 * stride];
460 v[3] = from[3 * stride];
461 return v;
462}
463
464template <>
465EIGEN_DEVICE_FUNC inline void pscatter<float, Packet4f>(float* to, const Packet4f& from, Index stride) {
466 EIGEN_MSA_DEBUG;
467
468 *to = from[0];
469 to += stride;
470 *to = from[1];
471 to += stride;
472 *to = from[2];
473 to += stride;
474 *to = from[3];
475}
476
477template <>
478EIGEN_DEVICE_FUNC inline void pscatter<int32_t, Packet4i>(int32_t* to, const Packet4i& from, Index stride) {
479 EIGEN_MSA_DEBUG;
480
481 *to = from[0];
482 to += stride;
483 *to = from[1];
484 to += stride;
485 *to = from[2];
486 to += stride;
487 *to = from[3];
488}
489
490template <>
491EIGEN_STRONG_INLINE void prefetch<float>(const float* addr) {
492 EIGEN_MSA_DEBUG;
493
494 __builtin_prefetch(addr);
495}
496
497template <>
498EIGEN_STRONG_INLINE void prefetch<int32_t>(const int32_t* addr) {
499 EIGEN_MSA_DEBUG;
500
501 __builtin_prefetch(addr);
502}
503
504template <>
505EIGEN_STRONG_INLINE float pfirst<Packet4f>(const Packet4f& a) {
506 EIGEN_MSA_DEBUG;
507
508 return a[0];
509}
510
511template <>
512EIGEN_STRONG_INLINE int32_t pfirst<Packet4i>(const Packet4i& a) {
513 EIGEN_MSA_DEBUG;
514
515 return a[0];
516}
517
518template <>
519EIGEN_STRONG_INLINE Packet4f preverse(const Packet4f& a) {
520 EIGEN_MSA_DEBUG;
521
522 return (Packet4f)__builtin_msa_shf_w((v4i32)a, EIGEN_MSA_SHF_I8(3, 2, 1, 0));
523}
524
525template <>
526EIGEN_STRONG_INLINE Packet4i preverse(const Packet4i& a) {
527 EIGEN_MSA_DEBUG;
528
529 return __builtin_msa_shf_w(a, EIGEN_MSA_SHF_I8(3, 2, 1, 0));
530}
531
532template <>
533EIGEN_STRONG_INLINE Packet4f pabs(const Packet4f& a) {
534 EIGEN_MSA_DEBUG;
535
536 return (Packet4f)__builtin_msa_bclri_w((v4u32)a, 31);
537}
538
539template <>
540EIGEN_STRONG_INLINE Packet4i pabs(const Packet4i& a) {
541 EIGEN_MSA_DEBUG;
542
543 Packet4i zero = __builtin_msa_ldi_w(0);
544 return __builtin_msa_add_a_w(zero, a);
545}
546
547template <>
548EIGEN_STRONG_INLINE float predux<Packet4f>(const Packet4f& a) {
549 EIGEN_MSA_DEBUG;
550
551 Packet4f s = padd(a, (Packet4f)__builtin_msa_shf_w((v4i32)a, EIGEN_MSA_SHF_I8(2, 3, 0, 1)));
552 s = padd(s, (Packet4f)__builtin_msa_shf_w((v4i32)s, EIGEN_MSA_SHF_I8(1, 0, 3, 2)));
553 return s[0];
554}
555
556template <>
557EIGEN_STRONG_INLINE bool predux_any(const Packet4f& a) {
558 return __builtin_msa_bnz_v((v16u8)a);
559}
560
561template <>
562EIGEN_STRONG_INLINE int32_t predux<Packet4i>(const Packet4i& a) {
563 EIGEN_MSA_DEBUG;
564
565 Packet4i s = padd(a, __builtin_msa_shf_w(a, EIGEN_MSA_SHF_I8(2, 3, 0, 1)));
566 s = padd(s, __builtin_msa_shf_w(s, EIGEN_MSA_SHF_I8(1, 0, 3, 2)));
567 return s[0];
568}
569
570template <>
571EIGEN_STRONG_INLINE bool predux_any(const Packet4i& a) {
572 return __builtin_msa_bnz_v((v16u8)a);
573}
574
575// Other reduction functions:
576// mul
577template <>
578EIGEN_STRONG_INLINE float predux_mul<Packet4f>(const Packet4f& a) {
579 EIGEN_MSA_DEBUG;
580
581 Packet4f p = pmul(a, (Packet4f)__builtin_msa_shf_w((v4i32)a, EIGEN_MSA_SHF_I8(2, 3, 0, 1)));
582 p = pmul(p, (Packet4f)__builtin_msa_shf_w((v4i32)p, EIGEN_MSA_SHF_I8(1, 0, 3, 2)));
583 return p[0];
584}
585
586template <>
587EIGEN_STRONG_INLINE int32_t predux_mul<Packet4i>(const Packet4i& a) {
588 EIGEN_MSA_DEBUG;
589
590 Packet4i p = pmul(a, __builtin_msa_shf_w(a, EIGEN_MSA_SHF_I8(2, 3, 0, 1)));
591 p = pmul(p, __builtin_msa_shf_w(p, EIGEN_MSA_SHF_I8(1, 0, 3, 2)));
592 return p[0];
593}
594
595// min
596template <>
597EIGEN_STRONG_INLINE float predux_min<Packet4f>(const Packet4f& a) {
598 EIGEN_MSA_DEBUG;
599
600 // Swap 64-bit halves of a.
601 Packet4f swapped = (Packet4f)__builtin_msa_shf_w((Packet4i)a, EIGEN_MSA_SHF_I8(2, 3, 0, 1));
602#if !EIGEN_FAST_MATH
603 // Detect presence of NaNs from pairs a[0]-a[2] and a[1]-a[3] as two 32-bit
604 // masks of all zeroes/ones in low 64 bits.
605 v16u8 unord = (v16u8)__builtin_msa_fcun_w(a, swapped);
606 // Combine the two masks into one: 64 ones if no NaNs, otherwise 64 zeroes.
607 unord = (v16u8)__builtin_msa_ceqi_d((v2i64)unord, 0);
608#endif
609 // Continue with min computation.
610 Packet4f v = __builtin_msa_fmin_w(a, swapped);
611 v = __builtin_msa_fmin_w(v, (Packet4f)__builtin_msa_shf_w((Packet4i)v, EIGEN_MSA_SHF_I8(1, 0, 3, 2)));
612#if !EIGEN_FAST_MATH
613 // Based on the mask select between v and 4 qNaNs.
614 v16u8 qnans = (v16u8)__builtin_msa_fill_w(0x7FC00000);
615 v = (Packet4f)__builtin_msa_bsel_v(unord, qnans, (v16u8)v);
616#endif
617 return v[0];
618}
619
620template <>
621EIGEN_STRONG_INLINE int32_t predux_min<Packet4i>(const Packet4i& a) {
622 EIGEN_MSA_DEBUG;
623
624 Packet4i m = pmin(a, __builtin_msa_shf_w(a, EIGEN_MSA_SHF_I8(2, 3, 0, 1)));
625 m = pmin(m, __builtin_msa_shf_w(m, EIGEN_MSA_SHF_I8(1, 0, 3, 2)));
626 return m[0];
627}
628
629// max
630template <>
631EIGEN_STRONG_INLINE float predux_max<Packet4f>(const Packet4f& a) {
632 EIGEN_MSA_DEBUG;
633
634 // Swap 64-bit halves of a.
635 Packet4f swapped = (Packet4f)__builtin_msa_shf_w((Packet4i)a, EIGEN_MSA_SHF_I8(2, 3, 0, 1));
636#if !EIGEN_FAST_MATH
637 // Detect presence of NaNs from pairs a[0]-a[2] and a[1]-a[3] as two 32-bit
638 // masks of all zeroes/ones in low 64 bits.
639 v16u8 unord = (v16u8)__builtin_msa_fcun_w(a, swapped);
640 // Combine the two masks into one: 64 ones if no NaNs, otherwise 64 zeroes.
641 unord = (v16u8)__builtin_msa_ceqi_d((v2i64)unord, 0);
642#endif
643 // Continue with max computation.
644 Packet4f v = __builtin_msa_fmax_w(a, swapped);
645 v = __builtin_msa_fmax_w(v, (Packet4f)__builtin_msa_shf_w((Packet4i)v, EIGEN_MSA_SHF_I8(1, 0, 3, 2)));
646#if !EIGEN_FAST_MATH
647 // Based on the mask select between v and 4 qNaNs.
648 v16u8 qnans = (v16u8)__builtin_msa_fill_w(0x7FC00000);
649 v = (Packet4f)__builtin_msa_bsel_v(unord, qnans, (v16u8)v);
650#endif
651 return v[0];
652}
653
654template <>
655EIGEN_STRONG_INLINE int32_t predux_max<Packet4i>(const Packet4i& a) {
656 EIGEN_MSA_DEBUG;
657
658 Packet4i m = pmax(a, __builtin_msa_shf_w(a, EIGEN_MSA_SHF_I8(2, 3, 0, 1)));
659 m = pmax(m, __builtin_msa_shf_w(m, EIGEN_MSA_SHF_I8(1, 0, 3, 2)));
660 return m[0];
661}
662
663inline std::ostream& operator<<(std::ostream& os, const PacketBlock<Packet4f, 4>& value) {
664 os << "[ " << value.packet[0] << "," << std::endl
665 << " " << value.packet[1] << "," << std::endl
666 << " " << value.packet[2] << "," << std::endl
667 << " " << value.packet[3] << " ]";
668 return os;
669}
670
671EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock<Packet4f, 4>& kernel) {
672 EIGEN_MSA_DEBUG;
673
674 v4i32 tmp1, tmp2, tmp3, tmp4;
675
676 tmp1 = __builtin_msa_ilvr_w((v4i32)kernel.packet[1], (v4i32)kernel.packet[0]);
677 tmp2 = __builtin_msa_ilvr_w((v4i32)kernel.packet[3], (v4i32)kernel.packet[2]);
678 tmp3 = __builtin_msa_ilvl_w((v4i32)kernel.packet[1], (v4i32)kernel.packet[0]);
679 tmp4 = __builtin_msa_ilvl_w((v4i32)kernel.packet[3], (v4i32)kernel.packet[2]);
680
681 kernel.packet[0] = (Packet4f)__builtin_msa_ilvr_d((v2i64)tmp2, (v2i64)tmp1);
682 kernel.packet[1] = (Packet4f)__builtin_msa_ilvod_d((v2i64)tmp2, (v2i64)tmp1);
683 kernel.packet[2] = (Packet4f)__builtin_msa_ilvr_d((v2i64)tmp4, (v2i64)tmp3);
684 kernel.packet[3] = (Packet4f)__builtin_msa_ilvod_d((v2i64)tmp4, (v2i64)tmp3);
685}
686
687inline std::ostream& operator<<(std::ostream& os, const PacketBlock<Packet4i, 4>& value) {
688 os << "[ " << value.packet[0] << "," << std::endl
689 << " " << value.packet[1] << "," << std::endl
690 << " " << value.packet[2] << "," << std::endl
691 << " " << value.packet[3] << " ]";
692 return os;
693}
694
695EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock<Packet4i, 4>& kernel) {
696 EIGEN_MSA_DEBUG;
697
698 v4i32 tmp1, tmp2, tmp3, tmp4;
699
700 tmp1 = __builtin_msa_ilvr_w(kernel.packet[1], kernel.packet[0]);
701 tmp2 = __builtin_msa_ilvr_w(kernel.packet[3], kernel.packet[2]);
702 tmp3 = __builtin_msa_ilvl_w(kernel.packet[1], kernel.packet[0]);
703 tmp4 = __builtin_msa_ilvl_w(kernel.packet[3], kernel.packet[2]);
704
705 kernel.packet[0] = (Packet4i)__builtin_msa_ilvr_d((v2i64)tmp2, (v2i64)tmp1);
706 kernel.packet[1] = (Packet4i)__builtin_msa_ilvod_d((v2i64)tmp2, (v2i64)tmp1);
707 kernel.packet[2] = (Packet4i)__builtin_msa_ilvr_d((v2i64)tmp4, (v2i64)tmp3);
708 kernel.packet[3] = (Packet4i)__builtin_msa_ilvod_d((v2i64)tmp4, (v2i64)tmp3);
709}
710
711template <>
712EIGEN_STRONG_INLINE Packet4f psqrt(const Packet4f& a) {
713 EIGEN_MSA_DEBUG;
714
715 return __builtin_msa_fsqrt_w(a);
716}
717
718template <>
719EIGEN_STRONG_INLINE Packet4f prsqrt(const Packet4f& a) {
720 EIGEN_MSA_DEBUG;
721
722#if EIGEN_FAST_MATH
723 return __builtin_msa_frsqrt_w(a);
724#else
725 Packet4f ones = __builtin_msa_ffint_s_w(__builtin_msa_ldi_w(1));
726 return pdiv(ones, psqrt(a));
727#endif
728}
729
730template <>
731EIGEN_STRONG_INLINE Packet4f pfloor<Packet4f>(const Packet4f& a) {
732 Packet4f v = a;
733 int32_t old_mode, new_mode;
734 asm volatile(
735 "cfcmsa %[old_mode], $1\n"
736 "ori %[new_mode], %[old_mode], 3\n" // 3 = round towards -INFINITY.
737 "ctcmsa $1, %[new_mode]\n"
738 "frint.w %w[v], %w[v]\n"
739 "ctcmsa $1, %[old_mode]\n"
740 : // outputs
741 [old_mode] "=r"(old_mode), [new_mode] "=r"(new_mode),
742 [v] "+f"(v)
743 : // inputs
744 : // clobbers
745 );
746 return v;
747}
748
749template <>
750EIGEN_STRONG_INLINE Packet4f pceil<Packet4f>(const Packet4f& a) {
751 Packet4f v = a;
752 int32_t old_mode, new_mode;
753 asm volatile(
754 "cfcmsa %[old_mode], $1\n"
755 "ori %[new_mode], %[old_mode], 3\n"
756 "xori %[new_mode], %[new_mode], 1\n" // 2 = round towards +INFINITY.
757 "ctcmsa $1, %[new_mode]\n"
758 "frint.w %w[v], %w[v]\n"
759 "ctcmsa $1, %[old_mode]\n"
760 : // outputs
761 [old_mode] "=r"(old_mode), [new_mode] "=r"(new_mode),
762 [v] "+f"(v)
763 : // inputs
764 : // clobbers
765 );
766 return v;
767}
768
769template <>
770EIGEN_STRONG_INLINE Packet4f pround<Packet4f>(const Packet4f& a) {
771 Packet4f v = a;
772 int32_t old_mode, new_mode;
773 asm volatile(
774 "cfcmsa %[old_mode], $1\n"
775 "ori %[new_mode], %[old_mode], 3\n"
776 "xori %[new_mode], %[new_mode], 3\n" // 0 = round to nearest, ties to even.
777 "ctcmsa $1, %[new_mode]\n"
778 "frint.w %w[v], %w[v]\n"
779 "ctcmsa $1, %[old_mode]\n"
780 : // outputs
781 [old_mode] "=r"(old_mode), [new_mode] "=r"(new_mode),
782 [v] "+f"(v)
783 : // inputs
784 : // clobbers
785 );
786 return v;
787}
788
789template <>
790EIGEN_STRONG_INLINE Packet4f print<Packet4f>(const Packet4f& a) {
791 // frint.w uses the current rounding mode (default: round to nearest, ties to even).
792 Packet4f v = a;
793 asm volatile("frint.w %w[v], %w[v]\n" : [v] "+f"(v));
794 return v;
795}
796
797template <>
798EIGEN_STRONG_INLINE Packet4f ptrunc<Packet4f>(const Packet4f& a) {
799 Packet4f v = a;
800 int32_t old_mode, new_mode;
801 asm volatile(
802 "cfcmsa %[old_mode], $1\n"
803 "ori %[new_mode], %[old_mode], 3\n"
804 "xori %[new_mode], %[new_mode], 2\n" // 1 = round toward zero.
805 "ctcmsa $1, %[new_mode]\n"
806 "frint.w %w[v], %w[v]\n"
807 "ctcmsa $1, %[old_mode]\n"
808 : // outputs
809 [old_mode] "=r"(old_mode), [new_mode] "=r"(new_mode),
810 [v] "+f"(v)
811 : // inputs
812 : // clobbers
813 );
814 return v;
815}
816
817template <>
818EIGEN_STRONG_INLINE Packet4f pcmp_lt_or_nan<Packet4f>(const Packet4f& a, const Packet4f& b) {
819 return (Packet4f)__builtin_msa_fcult_w(a, b);
820}
821
822//---------- double ----------
823
824typedef v2f64 Packet2d;
825typedef v2i64 Packet2l;
826typedef v2u64 Packet2ul;
827
828#define EIGEN_DECLARE_CONST_Packet2d(NAME, X) const Packet2d p2d_##NAME = {X, X}
829#define EIGEN_DECLARE_CONST_Packet2l(NAME, X) const Packet2l p2l_##NAME = {X, X}
830#define EIGEN_DECLARE_CONST_Packet2ul(NAME, X) const Packet2ul p2ul_##NAME = {X, X}
831
832inline std::ostream& operator<<(std::ostream& os, const Packet2d& value) {
833 os << "[ " << value[0] << ", " << value[1] << " ]";
834 return os;
835}
836
837inline std::ostream& operator<<(std::ostream& os, const Packet2l& value) {
838 os << "[ " << value[0] << ", " << value[1] << " ]";
839 return os;
840}
841
842inline std::ostream& operator<<(std::ostream& os, const Packet2ul& value) {
843 os << "[ " << value[0] << ", " << value[1] << " ]";
844 return os;
845}
846
847template <>
848struct packet_traits<double> : default_packet_traits {
849 typedef Packet2d type;
850 typedef Packet2d half;
851 enum {
852 Vectorizable = 1,
853 AlignedOnScalar = 1,
854 size = 2,
855 // FIXME: verify the Has* flags.
856 HasDiv = 1,
857 HasExp = 1,
858 HasSqrt = 1,
859 HasRsqrt = 1,
860 };
861};
862
863template <>
864struct unpacket_traits<Packet2d> {
865 typedef double type;
866 enum {
867 size = 2,
868 alignment = Aligned16,
869 vectorizable = true,
870 masked_load_available = false,
871 masked_store_available = false
872 };
873 typedef Packet2d half;
874};
875
876template <>
877EIGEN_STRONG_INLINE Packet2d pset1<Packet2d>(const double& from) {
878 EIGEN_MSA_DEBUG;
879
880 Packet2d value = {from, from};
881 return value;
882}
883
884template <>
885EIGEN_STRONG_INLINE Packet2d padd<Packet2d>(const Packet2d& a, const Packet2d& b) {
886 EIGEN_MSA_DEBUG;
887
888 return __builtin_msa_fadd_d(a, b);
889}
890
891template <>
892EIGEN_STRONG_INLINE Packet2d plset<Packet2d>(const double& a) {
893 EIGEN_MSA_DEBUG;
894
895 static const Packet2d countdown = {0.0, 1.0};
896 return padd(pset1<Packet2d>(a), countdown);
897}
898
899template <>
900EIGEN_STRONG_INLINE Packet2d psub<Packet2d>(const Packet2d& a, const Packet2d& b) {
901 EIGEN_MSA_DEBUG;
902
903 return __builtin_msa_fsub_d(a, b);
904}
905
906template <>
907EIGEN_STRONG_INLINE Packet2d pnegate(const Packet2d& a) {
908 EIGEN_MSA_DEBUG;
909
910 return (Packet2d)__builtin_msa_bnegi_d((v2u64)a, 63);
911}
912
913template <>
914EIGEN_STRONG_INLINE Packet2d pmul<Packet2d>(const Packet2d& a, const Packet2d& b) {
915 EIGEN_MSA_DEBUG;
916
917 return __builtin_msa_fmul_d(a, b);
918}
919
920template <>
921EIGEN_STRONG_INLINE Packet2d pdiv<Packet2d>(const Packet2d& a, const Packet2d& b) {
922 EIGEN_MSA_DEBUG;
923
924 return __builtin_msa_fdiv_d(a, b);
925}
926
927template <>
928EIGEN_STRONG_INLINE Packet2d pmadd(const Packet2d& a, const Packet2d& b, const Packet2d& c) {
929 EIGEN_MSA_DEBUG;
930
931 return __builtin_msa_fmadd_d(c, a, b);
932}
933
934// Logical Operations are not supported for float, so we have to reinterpret casts using MSA
935// intrinsics
936template <>
937EIGEN_STRONG_INLINE Packet2d pand<Packet2d>(const Packet2d& a, const Packet2d& b) {
938 EIGEN_MSA_DEBUG;
939
940 return (Packet2d)__builtin_msa_and_v((v16u8)a, (v16u8)b);
941}
942
943template <>
944EIGEN_STRONG_INLINE Packet2d por<Packet2d>(const Packet2d& a, const Packet2d& b) {
945 EIGEN_MSA_DEBUG;
946
947 return (Packet2d)__builtin_msa_or_v((v16u8)a, (v16u8)b);
948}
949
950template <>
951EIGEN_STRONG_INLINE Packet2d pxor<Packet2d>(const Packet2d& a, const Packet2d& b) {
952 EIGEN_MSA_DEBUG;
953
954 return (Packet2d)__builtin_msa_xor_v((v16u8)a, (v16u8)b);
955}
956
957template <>
958EIGEN_STRONG_INLINE Packet2d pandnot<Packet2d>(const Packet2d& a, const Packet2d& b) {
959 EIGEN_MSA_DEBUG;
960
961 return pand(a, (Packet2d)__builtin_msa_xori_b((v16u8)b, 255));
962}
963
964template <>
965EIGEN_STRONG_INLINE Packet2d pload<Packet2d>(const double* from) {
966 EIGEN_MSA_DEBUG;
967
968 EIGEN_DEBUG_UNALIGNED_LOAD return (Packet2d)__builtin_msa_ld_d(const_cast<double*>(from), 0);
969}
970
971template <>
972EIGEN_STRONG_INLINE Packet2d pmin<Packet2d>(const Packet2d& a, const Packet2d& b) {
973 EIGEN_MSA_DEBUG;
974
975#if EIGEN_FAST_MATH
976 // This prefers numbers to NaNs.
977 return __builtin_msa_fmin_d(a, b);
978#else
979 // This prefers NaNs to numbers.
980 v2i64 aNaN = __builtin_msa_fcun_d(a, a);
981 v2i64 aMinOrNaN = por(__builtin_msa_fclt_d(a, b), aNaN);
982 return (Packet2d)__builtin_msa_bsel_v((v16u8)aMinOrNaN, (v16u8)b, (v16u8)a);
983#endif
984}
985
986template <>
987EIGEN_STRONG_INLINE Packet2d pmax<Packet2d>(const Packet2d& a, const Packet2d& b) {
988 EIGEN_MSA_DEBUG;
989
990#if EIGEN_FAST_MATH
991 // This prefers numbers to NaNs.
992 return __builtin_msa_fmax_d(a, b);
993#else
994 // This prefers NaNs to numbers.
995 v2i64 aNaN = __builtin_msa_fcun_d(a, a);
996 v2i64 aMaxOrNaN = por(__builtin_msa_fclt_d(b, a), aNaN);
997 return (Packet2d)__builtin_msa_bsel_v((v16u8)aMaxOrNaN, (v16u8)b, (v16u8)a);
998#endif
999}
1000
1001template <>
1002EIGEN_STRONG_INLINE Packet2d ploadu<Packet2d>(const double* from) {
1003 EIGEN_MSA_DEBUG;
1004
1005 EIGEN_DEBUG_UNALIGNED_LOAD return (Packet2d)__builtin_msa_ld_d(const_cast<double*>(from), 0);
1006}
1007
1008template <>
1009EIGEN_STRONG_INLINE Packet2d ploaddup<Packet2d>(const double* from) {
1010 EIGEN_MSA_DEBUG;
1011
1012 Packet2d value = {*from, *from};
1013 return value;
1014}
1015
1016template <>
1017EIGEN_STRONG_INLINE void pstore<double>(double* to, const Packet2d& from) {
1018 EIGEN_MSA_DEBUG;
1019
1020 EIGEN_DEBUG_ALIGNED_STORE __builtin_msa_st_d((v2i64)from, to, 0);
1021}
1022
1023template <>
1024EIGEN_STRONG_INLINE void pstoreu<double>(double* to, const Packet2d& from) {
1025 EIGEN_MSA_DEBUG;
1026
1027 EIGEN_DEBUG_UNALIGNED_STORE __builtin_msa_st_d((v2i64)from, to, 0);
1028}
1029
1030template <>
1031EIGEN_DEVICE_FUNC inline Packet2d pgather<double, Packet2d>(const double* from, Index stride) {
1032 EIGEN_MSA_DEBUG;
1033
1034 Packet2d value;
1035 value[0] = *from;
1036 from += stride;
1037 value[1] = *from;
1038 return value;
1039}
1040
1041template <>
1042EIGEN_DEVICE_FUNC inline void pscatter<double, Packet2d>(double* to, const Packet2d& from, Index stride) {
1043 EIGEN_MSA_DEBUG;
1044
1045 *to = from[0];
1046 to += stride;
1047 *to = from[1];
1048}
1049
1050template <>
1051EIGEN_STRONG_INLINE void prefetch<double>(const double* addr) {
1052 EIGEN_MSA_DEBUG;
1053
1054 __builtin_prefetch(addr);
1055}
1056
1057template <>
1058EIGEN_STRONG_INLINE double pfirst<Packet2d>(const Packet2d& a) {
1059 EIGEN_MSA_DEBUG;
1060
1061 return a[0];
1062}
1063
1064template <>
1065EIGEN_STRONG_INLINE Packet2d preverse(const Packet2d& a) {
1066 EIGEN_MSA_DEBUG;
1067
1068 return (Packet2d)__builtin_msa_shf_w((v4i32)a, EIGEN_MSA_SHF_I8(2, 3, 0, 1));
1069}
1070
1071template <>
1072EIGEN_STRONG_INLINE Packet2d pabs(const Packet2d& a) {
1073 EIGEN_MSA_DEBUG;
1074
1075 return (Packet2d)__builtin_msa_bclri_d((v2u64)a, 63);
1076}
1077
1078template <>
1079EIGEN_STRONG_INLINE double predux<Packet2d>(const Packet2d& a) {
1080 EIGEN_MSA_DEBUG;
1081
1082 Packet2d s = padd(a, preverse(a));
1083 return s[0];
1084}
1085
1086template <>
1087EIGEN_STRONG_INLINE bool predux_any(const Packet2d& a) {
1088 return __builtin_msa_bnz_v((v16u8)a);
1089}
1090
1091// Other reduction functions:
1092// mul
1093template <>
1094EIGEN_STRONG_INLINE double predux_mul<Packet2d>(const Packet2d& a) {
1095 EIGEN_MSA_DEBUG;
1096
1097 Packet2d p = pmul(a, preverse(a));
1098 return p[0];
1099}
1100
1101// min
1102template <>
1103EIGEN_STRONG_INLINE double predux_min<Packet2d>(const Packet2d& a) {
1104 EIGEN_MSA_DEBUG;
1105
1106#if EIGEN_FAST_MATH
1107 Packet2d swapped = (Packet2d)__builtin_msa_shf_w((Packet4i)a, EIGEN_MSA_SHF_I8(2, 3, 0, 1));
1108 Packet2d v = __builtin_msa_fmin_d(a, swapped);
1109 return v[0];
1110#else
1111 double a0 = a[0], a1 = a[1];
1112 return ((numext::isnan)(a0) || a0 < a1) ? a0 : a1;
1113#endif
1114}
1115
1116// max
1117template <>
1118EIGEN_STRONG_INLINE double predux_max<Packet2d>(const Packet2d& a) {
1119 EIGEN_MSA_DEBUG;
1120
1121#if EIGEN_FAST_MATH
1122 Packet2d swapped = (Packet2d)__builtin_msa_shf_w((Packet4i)a, EIGEN_MSA_SHF_I8(2, 3, 0, 1));
1123 Packet2d v = __builtin_msa_fmax_d(a, swapped);
1124 return v[0];
1125#else
1126 double a0 = a[0], a1 = a[1];
1127 return ((numext::isnan)(a0) || a0 > a1) ? a0 : a1;
1128#endif
1129}
1130
1131template <>
1132EIGEN_STRONG_INLINE Packet2d psqrt(const Packet2d& a) {
1133 EIGEN_MSA_DEBUG;
1134
1135 return __builtin_msa_fsqrt_d(a);
1136}
1137
1138template <>
1139EIGEN_STRONG_INLINE Packet2d prsqrt(const Packet2d& a) {
1140 EIGEN_MSA_DEBUG;
1141
1142#if EIGEN_FAST_MATH
1143 return __builtin_msa_frsqrt_d(a);
1144#else
1145 Packet2d ones = __builtin_msa_ffint_s_d(__builtin_msa_ldi_d(1));
1146 return pdiv(ones, psqrt(a));
1147#endif
1148}
1149
1150inline std::ostream& operator<<(std::ostream& os, const PacketBlock<Packet2d, 2>& value) {
1151 os << "[ " << value.packet[0] << "," << std::endl << " " << value.packet[1] << " ]";
1152 return os;
1153}
1154
1155EIGEN_DEVICE_FUNC inline void ptranspose(PacketBlock<Packet2d, 2>& kernel) {
1156 EIGEN_MSA_DEBUG;
1157
1158 Packet2d trn1 = (Packet2d)__builtin_msa_ilvev_d((v2i64)kernel.packet[1], (v2i64)kernel.packet[0]);
1159 Packet2d trn2 = (Packet2d)__builtin_msa_ilvod_d((v2i64)kernel.packet[1], (v2i64)kernel.packet[0]);
1160 kernel.packet[0] = trn1;
1161 kernel.packet[1] = trn2;
1162}
1163
1164template <>
1165EIGEN_STRONG_INLINE Packet2d pfloor<Packet2d>(const Packet2d& a) {
1166 Packet2d v = a;
1167 int32_t old_mode, new_mode;
1168 asm volatile(
1169 "cfcmsa %[old_mode], $1\n"
1170 "ori %[new_mode], %[old_mode], 3\n" // 3 = round towards -INFINITY.
1171 "ctcmsa $1, %[new_mode]\n"
1172 "frint.d %w[v], %w[v]\n"
1173 "ctcmsa $1, %[old_mode]\n"
1174 : // outputs
1175 [old_mode] "=r"(old_mode), [new_mode] "=r"(new_mode),
1176 [v] "+f"(v)
1177 : // inputs
1178 : // clobbers
1179 );
1180 return v;
1181}
1182
1183template <>
1184EIGEN_STRONG_INLINE Packet2d pceil<Packet2d>(const Packet2d& a) {
1185 Packet2d v = a;
1186 int32_t old_mode, new_mode;
1187 asm volatile(
1188 "cfcmsa %[old_mode], $1\n"
1189 "ori %[new_mode], %[old_mode], 3\n"
1190 "xori %[new_mode], %[new_mode], 1\n" // 2 = round towards +INFINITY.
1191 "ctcmsa $1, %[new_mode]\n"
1192 "frint.d %w[v], %w[v]\n"
1193 "ctcmsa $1, %[old_mode]\n"
1194 : // outputs
1195 [old_mode] "=r"(old_mode), [new_mode] "=r"(new_mode),
1196 [v] "+f"(v)
1197 : // inputs
1198 : // clobbers
1199 );
1200 return v;
1201}
1202
1203template <>
1204EIGEN_STRONG_INLINE Packet2d pround<Packet2d>(const Packet2d& a) {
1205 Packet2d v = a;
1206 int32_t old_mode, new_mode;
1207 asm volatile(
1208 "cfcmsa %[old_mode], $1\n"
1209 "ori %[new_mode], %[old_mode], 3\n"
1210 "xori %[new_mode], %[new_mode], 3\n" // 0 = round to nearest, ties to even.
1211 "ctcmsa $1, %[new_mode]\n"
1212 "frint.d %w[v], %w[v]\n"
1213 "ctcmsa $1, %[old_mode]\n"
1214 : // outputs
1215 [old_mode] "=r"(old_mode), [new_mode] "=r"(new_mode),
1216 [v] "+f"(v)
1217 : // inputs
1218 : // clobbers
1219 );
1220 return v;
1221}
1222
1223template <>
1224EIGEN_STRONG_INLINE Packet2d print<Packet2d>(const Packet2d& a) {
1225 // frint.d uses the current rounding mode (default: round to nearest, ties to even).
1226 Packet2d v = a;
1227 asm volatile("frint.d %w[v], %w[v]\n" : [v] "+f"(v));
1228 return v;
1229}
1230
1231template <>
1232EIGEN_STRONG_INLINE Packet2d ptrunc<Packet2d>(const Packet2d& a) {
1233 Packet2d v = a;
1234 int32_t old_mode, new_mode;
1235 asm volatile(
1236 "cfcmsa %[old_mode], $1\n"
1237 "ori %[new_mode], %[old_mode], 3\n"
1238 "xori %[new_mode], %[new_mode], 2\n" // 1 = round toward zero.
1239 "ctcmsa $1, %[new_mode]\n"
1240 "frint.d %w[v], %w[v]\n"
1241 "ctcmsa $1, %[old_mode]\n"
1242 : // outputs
1243 [old_mode] "=r"(old_mode), [new_mode] "=r"(new_mode),
1244 [v] "+f"(v)
1245 : // inputs
1246 : // clobbers
1247 );
1248 return v;
1249}
1250
1251template <>
1252EIGEN_STRONG_INLINE Packet2d pcmp_lt_or_nan<Packet2d>(const Packet2d& a, const Packet2d& b) {
1253 return (Packet2d)__builtin_msa_fcult_d(a, b);
1254}
1255
1256} // end namespace internal
1257
1258} // end namespace Eigen
1259
1260#endif // EIGEN_PACKET_MATH_MSA_H
@ Aligned16
Definition Constants.h:238