Eigen  5.0.1
 
Loading...
Searching...
No Matches
Reductions.h
1// This file is part of Eigen, a lightweight C++ template library
2// for linear algebra.
3//
4// Copyright (C) 2025 Charlie Schlosser <cs.schlosser@gmail.com>
5//
6// This Source Code Form is subject to the terms of the Mozilla
7// Public License v. 2.0. If a copy of the MPL was not distributed
8// with this file, You can obtain one at http://mozilla.org/MPL/2.0/.
9// SPDX-License-Identifier: MPL-2.0
10
11#ifndef EIGEN_REDUCTIONS_SSE_H
12#define EIGEN_REDUCTIONS_SSE_H
13
14// IWYU pragma: private
15#include "../../InternalHeaderCheck.h"
16
17namespace Eigen {
18
19namespace internal {
20
21// Lane 0 of the result is lane 1 of a; the other lanes are unspecified. That is all
22// the 4->2->1 reductions below need from it, so take whichever single instruction the
23// target offers.
24EIGEN_STRONG_INLINE Packet4f sse_lane1(const Packet4f& a) {
25#ifdef EIGEN_VECTORIZE_SSE3
26 return _mm_movehdup_ps(a);
27#else
28 return _mm_shuffle_ps(a, a, 1);
29#endif
30}
31
32/* -- -- -- -- -- -- -- -- -- -- -- -- Packet16b -- -- -- -- -- -- -- -- -- -- -- -- */
33
34// Packet16b stores one bool per byte. Reduce the byte-wise zero mask rather than extracting and short-circuiting two
35// scalar halves. This also treats every nonzero byte as true, matching the scalar reduction for non-canonical inputs.
36template <>
37EIGEN_STRONG_INLINE Index predux_count(const Packet16b& a) {
38 const __m128i normalized = _mm_min_epu8(a, _mm_set1_epi8(1));
39 const __m128i sums = _mm_sad_epu8(normalized, _mm_setzero_si128());
40 return static_cast<Index>(_mm_cvtsi128_si32(sums)) +
41 static_cast<Index>(_mm_cvtsi128_si32(_mm_unpackhi_epi64(sums, sums)));
42}
43
44template <>
45EIGEN_STRONG_INLINE bool predux(const Packet16b& a) {
46 return _mm_movemask_epi8(_mm_cmpeq_epi8(a, _mm_setzero_si128())) != 0xffff;
47}
48
49template <>
50EIGEN_STRONG_INLINE bool predux_mul(const Packet16b& a) {
51 return _mm_movemask_epi8(_mm_cmpeq_epi8(a, _mm_setzero_si128())) == 0;
52}
53
54template <>
55EIGEN_STRONG_INLINE bool predux_min(const Packet16b& a) {
56 return predux_mul(a);
57}
58
59template <>
60EIGEN_STRONG_INLINE bool predux_max(const Packet16b& a) {
61 return predux(a);
62}
63
64template <>
65EIGEN_STRONG_INLINE bool predux_any(const Packet16b& a) {
66 return predux(a);
67}
68
69/* -- -- -- -- -- -- -- -- -- -- -- -- Packet4i -- -- -- -- -- -- -- -- -- -- -- -- */
70
71template <>
72EIGEN_STRONG_INLINE int predux(const Packet4i& a) {
73 Packet4i tmp = _mm_add_epi32(a, _mm_shuffle_epi32(a, _MM_SHUFFLE(0, 1, 2, 3)));
74 tmp = _mm_add_epi32(tmp, _mm_unpackhi_epi32(tmp, tmp));
75 return _mm_cvtsi128_si32(tmp);
76}
77
78template <>
79EIGEN_STRONG_INLINE int predux_mul(const Packet4i& a) {
80 Packet4i tmp = pmul<Packet4i>(a, _mm_shuffle_epi32(a, _MM_SHUFFLE(0, 1, 2, 3)));
81 tmp = pmul<Packet4i>(tmp, _mm_unpackhi_epi32(tmp, tmp));
82 return _mm_cvtsi128_si32(tmp);
83}
84
85#ifdef EIGEN_VECTORIZE_SSE4_1
86template <>
87EIGEN_STRONG_INLINE int predux_min(const Packet4i& a) {
88 Packet4i tmp = pmin<Packet4i>(a, _mm_shuffle_epi32(a, _MM_SHUFFLE(0, 1, 2, 3)));
89 tmp = pmin<Packet4i>(tmp, _mm_unpackhi_epi32(tmp, tmp));
90 return _mm_cvtsi128_si32(tmp);
91}
92
93template <>
94EIGEN_STRONG_INLINE int predux_max(const Packet4i& a) {
95 Packet4i tmp = pmax<Packet4i>(a, _mm_shuffle_epi32(a, _MM_SHUFFLE(0, 1, 2, 3)));
96 tmp = pmax<Packet4i>(tmp, _mm_unpackhi_epi32(tmp, tmp));
97 return _mm_cvtsi128_si32(tmp);
98}
99#endif
100
101template <>
102EIGEN_STRONG_INLINE bool predux_any(const Packet4i& a) {
103 return _mm_movemask_ps(_mm_castsi128_ps(a)) != 0x0;
104}
105
106/* -- -- -- -- -- -- -- -- -- -- -- -- Packet4ui -- -- -- -- -- -- -- -- -- -- -- -- */
107
108template <>
109EIGEN_STRONG_INLINE uint32_t predux(const Packet4ui& a) {
110 Packet4ui tmp = _mm_add_epi32(a, _mm_shuffle_epi32(a, _MM_SHUFFLE(0, 1, 2, 3)));
111 tmp = _mm_add_epi32(tmp, _mm_unpackhi_epi32(tmp, tmp));
112 return static_cast<uint32_t>(_mm_cvtsi128_si32(tmp));
113}
114
115template <>
116EIGEN_STRONG_INLINE uint32_t predux_mul(const Packet4ui& a) {
117 Packet4ui tmp = pmul<Packet4ui>(a, _mm_shuffle_epi32(a, _MM_SHUFFLE(0, 1, 2, 3)));
118 tmp = pmul<Packet4ui>(tmp, _mm_unpackhi_epi32(tmp, tmp));
119 return static_cast<uint32_t>(_mm_cvtsi128_si32(tmp));
120}
121
122#ifdef EIGEN_VECTORIZE_SSE4_1
123template <>
124EIGEN_STRONG_INLINE uint32_t predux_min(const Packet4ui& a) {
125 Packet4ui tmp = pmin<Packet4ui>(a, _mm_shuffle_epi32(a, _MM_SHUFFLE(0, 1, 2, 3)));
126 tmp = pmin<Packet4ui>(tmp, _mm_unpackhi_epi32(tmp, tmp));
127 return static_cast<uint32_t>(_mm_cvtsi128_si32(tmp));
128}
129
130template <>
131EIGEN_STRONG_INLINE uint32_t predux_max(const Packet4ui& a) {
132 Packet4ui tmp = pmax<Packet4ui>(a, _mm_shuffle_epi32(a, _MM_SHUFFLE(0, 1, 2, 3)));
133 tmp = pmax<Packet4ui>(tmp, _mm_unpackhi_epi32(tmp, tmp));
134 return static_cast<uint32_t>(_mm_cvtsi128_si32(tmp));
135}
136#endif
137
138template <>
139EIGEN_STRONG_INLINE bool predux_any(const Packet4ui& a) {
140 return _mm_movemask_ps(_mm_castsi128_ps(a)) != 0x0;
141}
142
143/* -- -- -- -- -- -- -- -- -- -- -- -- Packet2l -- -- -- -- -- -- -- -- -- -- -- -- */
144
145template <>
146EIGEN_STRONG_INLINE int64_t predux(const Packet2l& a) {
147 Packet2l tmp = _mm_add_epi64(a, _mm_unpackhi_epi64(a, a));
148 return pfirst(tmp);
149}
150
151template <>
152EIGEN_STRONG_INLINE bool predux_any(const Packet2l& a) {
153 return _mm_movemask_pd(_mm_castsi128_pd(a)) != 0x0;
154}
155
156/* -- -- -- -- -- -- -- -- -- -- -- -- Packet4f -- -- -- -- -- -- -- -- -- -- -- -- */
157
158template <>
159EIGEN_STRONG_INLINE float predux(const Packet4f& a) {
160 // The 2->1 step is the low-lane form, as for Packet2d. It reads an already-reduced
161 // temporary rather than the live packet, so unlike Packet2d it needs no encoding split.
162 Packet4f tmp = _mm_add_ps(a, _mm_movehl_ps(a, a));
163 tmp = _mm_add_ss(tmp, sse_lane1(tmp));
164 return _mm_cvtss_f32(tmp);
165}
166
167template <>
168EIGEN_STRONG_INLINE float predux_mul(const Packet4f& a) {
169 Packet4f tmp = _mm_mul_ps(a, _mm_movehl_ps(a, a));
170 tmp = _mm_mul_ss(tmp, sse_lane1(tmp));
171 return _mm_cvtss_f32(tmp);
172}
173
174template <>
175EIGEN_STRONG_INLINE float predux_min(const Packet4f& a) {
176 Packet4f tmp = pmin<Packet4f>(a, _mm_movehl_ps(a, a));
177 tmp = pmin<Packet4f>(tmp, sse_lane1(tmp));
178 return _mm_cvtss_f32(tmp);
179}
180
181template <>
182EIGEN_STRONG_INLINE float predux_min<PropagateNumbers>(const Packet4f& a) {
183 Packet4f tmp = pmin<PropagateNumbers, Packet4f>(a, _mm_movehl_ps(a, a));
184 tmp = pmin<PropagateNumbers, Packet4f>(tmp, sse_lane1(tmp));
185 return _mm_cvtss_f32(tmp);
186}
187
188template <>
189EIGEN_STRONG_INLINE float predux_min<PropagateNaN>(const Packet4f& a) {
190 Packet4f tmp = pmin<PropagateNaN, Packet4f>(a, _mm_movehl_ps(a, a));
191 tmp = pmin<PropagateNaN, Packet4f>(tmp, sse_lane1(tmp));
192 return _mm_cvtss_f32(tmp);
193}
194
195template <>
196EIGEN_STRONG_INLINE float predux_max(const Packet4f& a) {
197 Packet4f tmp = pmax<Packet4f>(a, _mm_movehl_ps(a, a));
198 tmp = pmax<Packet4f>(tmp, sse_lane1(tmp));
199 return _mm_cvtss_f32(tmp);
200}
201
202template <>
203EIGEN_STRONG_INLINE float predux_max<PropagateNumbers>(const Packet4f& a) {
204 Packet4f tmp = pmax<PropagateNumbers, Packet4f>(a, _mm_movehl_ps(a, a));
205 tmp = pmax<PropagateNumbers, Packet4f>(tmp, sse_lane1(tmp));
206 return _mm_cvtss_f32(tmp);
207}
208
209template <>
210EIGEN_STRONG_INLINE float predux_max<PropagateNaN>(const Packet4f& a) {
211 Packet4f tmp = pmax<PropagateNaN, Packet4f>(a, _mm_movehl_ps(a, a));
212 tmp = pmax<PropagateNaN, Packet4f>(tmp, sse_lane1(tmp));
213 return _mm_cvtss_f32(tmp);
214}
215
216template <>
217EIGEN_STRONG_INLINE bool predux_any(const Packet4f& a) {
218 return _mm_movemask_ps(a) != 0x0;
219}
220
221#ifdef EIGEN_VECTORIZE_SSE4_2
222template <>
223EIGEN_STRONG_INLINE Index predux_count(const Packet4f& a) {
224 const unsigned int mask = static_cast<unsigned int>(_mm_movemask_ps(_mm_cmpneq_ps(a, _mm_setzero_ps())));
225 return Index(popcount(mask));
226}
227#endif
228
229/* -- -- -- -- -- -- -- -- -- -- -- -- Packet2d -- -- -- -- -- -- -- -- -- -- -- -- */
230
231// The 2->1 step is not packed: a packed step pins the result in a vector register, so
232// neighbouring reductions -- one per coefficient of a small coeff-based product -- are
233// not re-packed into one store (~16% on clang/AVX2). Take the high lane first: pfirst(a)
234// is a's register, so reading it first keeps a live across the shuffle and gcc copies it
235// out. Without VEX the low-lane form wins when a is an accumulator rather than the data.
236template <>
237EIGEN_STRONG_INLINE double predux(const Packet2d& a) {
238#ifdef EIGEN_VECTORIZE_AVX
239 const double hi = pfirst(preverse(a));
240 return pfirst(a) + hi;
241#else
242 return _mm_cvtsd_f64(_mm_add_sd(a, preverse(a)));
243#endif
244}
245
246template <>
247EIGEN_STRONG_INLINE double predux_mul(const Packet2d& a) {
248#ifdef EIGEN_VECTORIZE_AVX
249 const double hi = pfirst(preverse(a));
250 return pfirst(a) * hi;
251#else
252 return _mm_cvtsd_f64(_mm_mul_sd(a, preverse(a)));
253#endif
254}
255
256template <>
257EIGEN_STRONG_INLINE double predux_min(const Packet2d& a) {
258#ifdef EIGEN_VECTORIZE_AVX
259 const double hi = pfirst(preverse(a));
260 return pmin<double>(pfirst(a), hi);
261#else
262 // _mm_unpackhi_pd, not preverse: clang folds this whole reduction into a scalar
263 // load from lane 1, and only recognises that spelling.
264 return _mm_cvtsd_f64(pmin<Packet2d>(a, _mm_unpackhi_pd(a, a)));
265#endif
266}
267
268template <>
269EIGEN_STRONG_INLINE double predux_min<PropagateNumbers>(const Packet2d& a) {
270 Packet2d tmp = pmin<PropagateNumbers, Packet2d>(a, _mm_unpackhi_pd(a, a));
271 return _mm_cvtsd_f64(tmp);
272}
273
274template <>
275EIGEN_STRONG_INLINE double predux_min<PropagateNaN>(const Packet2d& a) {
276 Packet2d tmp = pmin<PropagateNaN, Packet2d>(a, _mm_unpackhi_pd(a, a));
277 return _mm_cvtsd_f64(tmp);
278}
279
280template <>
281EIGEN_STRONG_INLINE double predux_max(const Packet2d& a) {
282#ifdef EIGEN_VECTORIZE_AVX
283 const double hi = pfirst(preverse(a));
284 return pmax<double>(pfirst(a), hi);
285#else
286 // _mm_unpackhi_pd, not preverse: clang folds this whole reduction into a scalar
287 // load from lane 1, and only recognises that spelling.
288 return _mm_cvtsd_f64(pmax<Packet2d>(a, _mm_unpackhi_pd(a, a)));
289#endif
290}
291
292template <>
293EIGEN_STRONG_INLINE double predux_max<PropagateNumbers>(const Packet2d& a) {
294 Packet2d tmp = pmax<PropagateNumbers, Packet2d>(a, _mm_unpackhi_pd(a, a));
295 return _mm_cvtsd_f64(tmp);
296}
297
298template <>
299EIGEN_STRONG_INLINE double predux_max<PropagateNaN>(const Packet2d& a) {
300 Packet2d tmp = pmax<PropagateNaN, Packet2d>(a, _mm_unpackhi_pd(a, a));
301 return _mm_cvtsd_f64(tmp);
302}
303
304template <>
305EIGEN_STRONG_INLINE bool predux_any(const Packet2d& a) {
306 return _mm_movemask_pd(a) != 0x0;
307}
308
309#ifdef EIGEN_VECTORIZE_SSE4_2
310template <>
311EIGEN_STRONG_INLINE Index predux_count(const Packet2d& a) {
312 const unsigned int mask = static_cast<unsigned int>(_mm_movemask_pd(_mm_cmpneq_pd(a, _mm_setzero_pd())));
313 return Index(popcount(mask));
314}
315#endif
316
317} // end namespace internal
318
319} // end namespace Eigen
320
321#endif // EIGEN_REDUCTIONS_SSE_H