11#ifndef EIGEN_REDUCTIONS_SSE_H
12#define EIGEN_REDUCTIONS_SSE_H
15#include "../../InternalHeaderCheck.h"
24EIGEN_STRONG_INLINE Packet4f sse_lane1(
const Packet4f& a) {
25#ifdef EIGEN_VECTORIZE_SSE3
26 return _mm_movehdup_ps(a);
28 return _mm_shuffle_ps(a, a, 1);
37EIGEN_STRONG_INLINE Index predux_count(
const Packet16b& a) {
38 const __m128i normalized = _mm_min_epu8(a, _mm_set1_epi8(1));
39 const __m128i sums = _mm_sad_epu8(normalized, _mm_setzero_si128());
40 return static_cast<Index
>(_mm_cvtsi128_si32(sums)) +
41 static_cast<Index
>(_mm_cvtsi128_si32(_mm_unpackhi_epi64(sums, sums)));
45EIGEN_STRONG_INLINE
bool predux(
const Packet16b& a) {
46 return _mm_movemask_epi8(_mm_cmpeq_epi8(a, _mm_setzero_si128())) != 0xffff;
50EIGEN_STRONG_INLINE
bool predux_mul(
const Packet16b& a) {
51 return _mm_movemask_epi8(_mm_cmpeq_epi8(a, _mm_setzero_si128())) == 0;
55EIGEN_STRONG_INLINE
bool predux_min(
const Packet16b& a) {
60EIGEN_STRONG_INLINE
bool predux_max(
const Packet16b& a) {
65EIGEN_STRONG_INLINE
bool predux_any(
const Packet16b& a) {
72EIGEN_STRONG_INLINE
int predux(
const Packet4i& a) {
73 Packet4i tmp = _mm_add_epi32(a, _mm_shuffle_epi32(a, _MM_SHUFFLE(0, 1, 2, 3)));
74 tmp = _mm_add_epi32(tmp, _mm_unpackhi_epi32(tmp, tmp));
75 return _mm_cvtsi128_si32(tmp);
79EIGEN_STRONG_INLINE
int predux_mul(
const Packet4i& a) {
80 Packet4i tmp = pmul<Packet4i>(a, _mm_shuffle_epi32(a, _MM_SHUFFLE(0, 1, 2, 3)));
81 tmp = pmul<Packet4i>(tmp, _mm_unpackhi_epi32(tmp, tmp));
82 return _mm_cvtsi128_si32(tmp);
85#ifdef EIGEN_VECTORIZE_SSE4_1
87EIGEN_STRONG_INLINE
int predux_min(
const Packet4i& a) {
88 Packet4i tmp = pmin<Packet4i>(a, _mm_shuffle_epi32(a, _MM_SHUFFLE(0, 1, 2, 3)));
89 tmp = pmin<Packet4i>(tmp, _mm_unpackhi_epi32(tmp, tmp));
90 return _mm_cvtsi128_si32(tmp);
94EIGEN_STRONG_INLINE
int predux_max(
const Packet4i& a) {
95 Packet4i tmp = pmax<Packet4i>(a, _mm_shuffle_epi32(a, _MM_SHUFFLE(0, 1, 2, 3)));
96 tmp = pmax<Packet4i>(tmp, _mm_unpackhi_epi32(tmp, tmp));
97 return _mm_cvtsi128_si32(tmp);
102EIGEN_STRONG_INLINE
bool predux_any(
const Packet4i& a) {
103 return _mm_movemask_ps(_mm_castsi128_ps(a)) != 0x0;
109EIGEN_STRONG_INLINE uint32_t predux(
const Packet4ui& a) {
110 Packet4ui tmp = _mm_add_epi32(a, _mm_shuffle_epi32(a, _MM_SHUFFLE(0, 1, 2, 3)));
111 tmp = _mm_add_epi32(tmp, _mm_unpackhi_epi32(tmp, tmp));
112 return static_cast<uint32_t
>(_mm_cvtsi128_si32(tmp));
116EIGEN_STRONG_INLINE uint32_t predux_mul(
const Packet4ui& a) {
117 Packet4ui tmp = pmul<Packet4ui>(a, _mm_shuffle_epi32(a, _MM_SHUFFLE(0, 1, 2, 3)));
118 tmp = pmul<Packet4ui>(tmp, _mm_unpackhi_epi32(tmp, tmp));
119 return static_cast<uint32_t
>(_mm_cvtsi128_si32(tmp));
122#ifdef EIGEN_VECTORIZE_SSE4_1
124EIGEN_STRONG_INLINE uint32_t predux_min(
const Packet4ui& a) {
125 Packet4ui tmp = pmin<Packet4ui>(a, _mm_shuffle_epi32(a, _MM_SHUFFLE(0, 1, 2, 3)));
126 tmp = pmin<Packet4ui>(tmp, _mm_unpackhi_epi32(tmp, tmp));
127 return static_cast<uint32_t
>(_mm_cvtsi128_si32(tmp));
131EIGEN_STRONG_INLINE uint32_t predux_max(
const Packet4ui& a) {
132 Packet4ui tmp = pmax<Packet4ui>(a, _mm_shuffle_epi32(a, _MM_SHUFFLE(0, 1, 2, 3)));
133 tmp = pmax<Packet4ui>(tmp, _mm_unpackhi_epi32(tmp, tmp));
134 return static_cast<uint32_t
>(_mm_cvtsi128_si32(tmp));
139EIGEN_STRONG_INLINE
bool predux_any(
const Packet4ui& a) {
140 return _mm_movemask_ps(_mm_castsi128_ps(a)) != 0x0;
146EIGEN_STRONG_INLINE int64_t predux(
const Packet2l& a) {
147 Packet2l tmp = _mm_add_epi64(a, _mm_unpackhi_epi64(a, a));
152EIGEN_STRONG_INLINE
bool predux_any(
const Packet2l& a) {
153 return _mm_movemask_pd(_mm_castsi128_pd(a)) != 0x0;
159EIGEN_STRONG_INLINE
float predux(
const Packet4f& a) {
162 Packet4f tmp = _mm_add_ps(a, _mm_movehl_ps(a, a));
163 tmp = _mm_add_ss(tmp, sse_lane1(tmp));
164 return _mm_cvtss_f32(tmp);
168EIGEN_STRONG_INLINE
float predux_mul(
const Packet4f& a) {
169 Packet4f tmp = _mm_mul_ps(a, _mm_movehl_ps(a, a));
170 tmp = _mm_mul_ss(tmp, sse_lane1(tmp));
171 return _mm_cvtss_f32(tmp);
175EIGEN_STRONG_INLINE
float predux_min(
const Packet4f& a) {
176 Packet4f tmp = pmin<Packet4f>(a, _mm_movehl_ps(a, a));
177 tmp = pmin<Packet4f>(tmp, sse_lane1(tmp));
178 return _mm_cvtss_f32(tmp);
182EIGEN_STRONG_INLINE
float predux_min<PropagateNumbers>(
const Packet4f& a) {
183 Packet4f tmp = pmin<PropagateNumbers, Packet4f>(a, _mm_movehl_ps(a, a));
184 tmp = pmin<PropagateNumbers, Packet4f>(tmp, sse_lane1(tmp));
185 return _mm_cvtss_f32(tmp);
189EIGEN_STRONG_INLINE
float predux_min<PropagateNaN>(
const Packet4f& a) {
190 Packet4f tmp = pmin<PropagateNaN, Packet4f>(a, _mm_movehl_ps(a, a));
191 tmp = pmin<PropagateNaN, Packet4f>(tmp, sse_lane1(tmp));
192 return _mm_cvtss_f32(tmp);
196EIGEN_STRONG_INLINE
float predux_max(
const Packet4f& a) {
197 Packet4f tmp = pmax<Packet4f>(a, _mm_movehl_ps(a, a));
198 tmp = pmax<Packet4f>(tmp, sse_lane1(tmp));
199 return _mm_cvtss_f32(tmp);
203EIGEN_STRONG_INLINE
float predux_max<PropagateNumbers>(
const Packet4f& a) {
204 Packet4f tmp = pmax<PropagateNumbers, Packet4f>(a, _mm_movehl_ps(a, a));
205 tmp = pmax<PropagateNumbers, Packet4f>(tmp, sse_lane1(tmp));
206 return _mm_cvtss_f32(tmp);
210EIGEN_STRONG_INLINE
float predux_max<PropagateNaN>(
const Packet4f& a) {
211 Packet4f tmp = pmax<PropagateNaN, Packet4f>(a, _mm_movehl_ps(a, a));
212 tmp = pmax<PropagateNaN, Packet4f>(tmp, sse_lane1(tmp));
213 return _mm_cvtss_f32(tmp);
217EIGEN_STRONG_INLINE
bool predux_any(
const Packet4f& a) {
218 return _mm_movemask_ps(a) != 0x0;
221#ifdef EIGEN_VECTORIZE_SSE4_2
223EIGEN_STRONG_INLINE Index predux_count(
const Packet4f& a) {
224 const unsigned int mask =
static_cast<unsigned int>(_mm_movemask_ps(_mm_cmpneq_ps(a, _mm_setzero_ps())));
225 return Index(popcount(mask));
237EIGEN_STRONG_INLINE
double predux(
const Packet2d& a) {
238#ifdef EIGEN_VECTORIZE_AVX
239 const double hi = pfirst(preverse(a));
240 return pfirst(a) + hi;
242 return _mm_cvtsd_f64(_mm_add_sd(a, preverse(a)));
247EIGEN_STRONG_INLINE
double predux_mul(
const Packet2d& a) {
248#ifdef EIGEN_VECTORIZE_AVX
249 const double hi = pfirst(preverse(a));
250 return pfirst(a) * hi;
252 return _mm_cvtsd_f64(_mm_mul_sd(a, preverse(a)));
257EIGEN_STRONG_INLINE
double predux_min(
const Packet2d& a) {
258#ifdef EIGEN_VECTORIZE_AVX
259 const double hi = pfirst(preverse(a));
260 return pmin<double>(pfirst(a), hi);
264 return _mm_cvtsd_f64(pmin<Packet2d>(a, _mm_unpackhi_pd(a, a)));
269EIGEN_STRONG_INLINE
double predux_min<PropagateNumbers>(
const Packet2d& a) {
270 Packet2d tmp = pmin<PropagateNumbers, Packet2d>(a, _mm_unpackhi_pd(a, a));
271 return _mm_cvtsd_f64(tmp);
275EIGEN_STRONG_INLINE
double predux_min<PropagateNaN>(
const Packet2d& a) {
276 Packet2d tmp = pmin<PropagateNaN, Packet2d>(a, _mm_unpackhi_pd(a, a));
277 return _mm_cvtsd_f64(tmp);
281EIGEN_STRONG_INLINE
double predux_max(
const Packet2d& a) {
282#ifdef EIGEN_VECTORIZE_AVX
283 const double hi = pfirst(preverse(a));
284 return pmax<double>(pfirst(a), hi);
288 return _mm_cvtsd_f64(pmax<Packet2d>(a, _mm_unpackhi_pd(a, a)));
293EIGEN_STRONG_INLINE
double predux_max<PropagateNumbers>(
const Packet2d& a) {
294 Packet2d tmp = pmax<PropagateNumbers, Packet2d>(a, _mm_unpackhi_pd(a, a));
295 return _mm_cvtsd_f64(tmp);
299EIGEN_STRONG_INLINE
double predux_max<PropagateNaN>(
const Packet2d& a) {
300 Packet2d tmp = pmax<PropagateNaN, Packet2d>(a, _mm_unpackhi_pd(a, a));
301 return _mm_cvtsd_f64(tmp);
305EIGEN_STRONG_INLINE
bool predux_any(
const Packet2d& a) {
306 return _mm_movemask_pd(a) != 0x0;
309#ifdef EIGEN_VECTORIZE_SSE4_2
311EIGEN_STRONG_INLINE Index predux_count(
const Packet2d& a) {
312 const unsigned int mask =
static_cast<unsigned int>(_mm_movemask_pd(_mm_cmpneq_pd(a, _mm_setzero_pd())));
313 return Index(popcount(mask));