Eigen  5.0.1
 
Loading...
Searching...
No Matches
Reductions.h
1// This file is part of Eigen, a lightweight C++ template library
2// for linear algebra.
3//
4// Copyright (C) 2025 Rasmus Munk Larsen
5//
6// This Source Code Form is subject to the terms of the Mozilla
7// Public License v. 2.0. If a copy of the MPL was not distributed
8// with this file, You can obtain one at http://mozilla.org/MPL/2.0/.
9// SPDX-License-Identifier: MPL-2.0
10
11#ifndef EIGEN_REDUCTIONS_CLANG_H
12#define EIGEN_REDUCTIONS_CLANG_H
13
14// IWYU pragma: private
15#include "../../InternalHeaderCheck.h"
16
17namespace Eigen {
18namespace internal {
19
20// --- Reductions ---
21// __builtin_reduce_{min,max} lower well for the integer packets only: for
22// floating point their strict NaN-ordering semantics scalarize into a serial
23// compare-blend chain, so PacketXf/PacketXd use the halving trees below.
24#if EIGEN_HAS_BUILTIN(__builtin_reduce_min) && EIGEN_HAS_BUILTIN(__builtin_reduce_max)
25#define EIGEN_CLANG_PACKET_REDUX_MINMAX_INT(PACKET_TYPE) \
26 template <> \
27 EIGEN_STRONG_INLINE unpacket_traits<PACKET_TYPE>::type predux_min(const PACKET_TYPE& a) { \
28 return __builtin_reduce_min(a); \
29 } \
30 template <> \
31 EIGEN_STRONG_INLINE unpacket_traits<PACKET_TYPE>::type predux_max(const PACKET_TYPE& a) { \
32 return __builtin_reduce_max(a); \
33 }
34
35EIGEN_CLANG_PACKET_REDUX_MINMAX_INT(PacketXi)
36EIGEN_CLANG_PACKET_REDUX_MINMAX_INT(PacketXl)
37#undef EIGEN_CLANG_PACKET_REDUX_MINMAX_INT
38#endif
39
40#if EIGEN_HAS_BUILTIN(__builtin_reduce_or)
41// Test the integer view: comparing an all-ones (NaN bit pattern) float mask
42// against zero is fair game for -ffast-math to fold away, as with pselect.
43#define EIGEN_CLANG_PACKET_REDUX_ANY(PACKET_TYPE) \
44 template <> \
45 EIGEN_STRONG_INLINE bool predux_any(const PACKET_TYPE& a) { \
46 return __builtin_reduce_or(reinterpret_cast<detail::signed_vector_t<PACKET_TYPE>>(a) != 0) != 0; \
47 }
48
49EIGEN_CLANG_PACKET_REDUX_ANY(PacketXf)
50EIGEN_CLANG_PACKET_REDUX_ANY(PacketXd)
51EIGEN_CLANG_PACKET_REDUX_ANY(PacketXi)
52EIGEN_CLANG_PACKET_REDUX_ANY(PacketXl)
53#undef EIGEN_CLANG_PACKET_REDUX_ANY
54#endif
55
56#if EIGEN_HAS_BUILTIN(__builtin_reduce_add) && EIGEN_HAS_BUILTIN(__builtin_reduce_mul)
57#define EIGEN_CLANG_PACKET_REDUX_INT(PACKET_TYPE) \
58 template <> \
59 EIGEN_STRONG_INLINE unpacket_traits<PACKET_TYPE>::type predux<PACKET_TYPE>(const PACKET_TYPE& a) { \
60 return __builtin_reduce_add(a); \
61 } \
62 template <> \
63 EIGEN_STRONG_INLINE unpacket_traits<PACKET_TYPE>::type predux_mul<PACKET_TYPE>(const PACKET_TYPE& a) { \
64 return __builtin_reduce_mul(a); \
65 }
66
67// __builtin_reduce_{mul,add} are only defined for integer types.
68EIGEN_CLANG_PACKET_REDUX_INT(PacketXi)
69EIGEN_CLANG_PACKET_REDUX_INT(PacketXl)
70#undef EIGEN_CLANG_PACKET_REDUX_INT
71#endif
72
73#if EIGEN_HAS_BUILTIN(__builtin_shufflevector)
74namespace detail {
75
76// Folds `a` in half with `op` until two elements are left and returns them as
77// (even, odd). Callers combine those two with the same operation; splitting the
78// final step out is what lets the complex reductions read off the accumulated
79// real and imaginary parts separately. The halves are shuffled inline rather
80// than through lower_half/upper_half: an 8-byte half returned by value is
81// ABI-lowered to a scalar double in IR even under forced inlining, and the
82// leftover bitcasts perturb LLVM's canonicalization of the commutative fold,
83// measurably changing the generated code.
84template <int N>
85struct halving_reduce {
86 template <typename VectorT, typename Op, std::size_t... Is>
87 static EIGEN_STRONG_INLINE scalar_pair_t<VectorT> fold(const VectorT& a, Op op, std::index_sequence<Is...>) {
88 return halving_reduce<N / 2>::run(
89 op(__builtin_shufflevector(a, a, Is...), __builtin_shufflevector(a, a, (sizeof...(Is) + Is)...)), op);
90 }
91
92 template <typename VectorT, typename Op>
93 static EIGEN_STRONG_INLINE scalar_pair_t<VectorT> run(const VectorT& a, Op op) {
94 return fold(a, op, std::make_index_sequence<N / 2>{});
95 }
96};
97
98template <>
99struct halving_reduce<2> {
100 template <typename VectorT, typename Op>
101 static EIGEN_STRONG_INLINE scalar_pair_t<VectorT> run(const VectorT& a, Op /*op*/) {
102 return {a[0], a[1]};
103 }
104};
105
106template <typename VectorT>
107EIGEN_STRONG_INLINE scalar_pair_t<VectorT> reduce_add_pairs(const VectorT& a) {
108 return halving_reduce<vector_elements<VectorT>()>::run(a, [](const auto& x, const auto& y) { return x + y; });
109}
110
111// Folds the packet all the way to a scalar with `op`, which must also be
112// applicable to bare scalars for the final step.
113template <typename VectorT, typename Op>
114EIGEN_STRONG_INLINE scalar_type_of_vector_t<VectorT> tree_reduce(const VectorT& a, Op op) {
115 const scalar_pair_t<VectorT> even_odd = halving_reduce<vector_elements<VectorT>()>::run(a, op);
116 return op(even_odd.first, even_odd.second);
117}
118
119// Multiplies the two halves of a complex packet until two complex values are
120// left, then multiplies those as scalars. Unlike the reductions above this
121// cannot work on the real vector, because complex multiplication mixes lanes.
122template <int N>
123struct complex_predux_mul {
124 template <typename RealScalar>
125 static EIGEN_STRONG_INLINE std::complex<RealScalar> run(const complex_packet_wrapper<RealScalar, N>& a) {
126 using HalfPacket = complex_packet_wrapper<RealScalar, N / 2>;
127 return complex_predux_mul<N / 2>::run(complex_pmul_impl(HalfPacket(lower_half(a.v)), HalfPacket(upper_half(a.v)),
128 complex_real_indices<HalfPacket>{}));
129 }
130};
131
132template <>
133struct complex_predux_mul<2> {
134 template <typename RealScalar>
135 static EIGEN_STRONG_INLINE std::complex<RealScalar> run(const complex_packet_wrapper<RealScalar, 2>& a) {
136 return a[0] * a[1];
137 }
138};
139
140template <>
141struct complex_predux_mul<1> {
142 template <typename RealScalar>
143 static EIGEN_STRONG_INLINE std::complex<RealScalar> run(const complex_packet_wrapper<RealScalar, 1>& a) {
144 return a[0];
145 }
146};
147
148} // namespace detail
149
150// --- predux and predux_mul for float and double ---
151// __builtin_reduce_{add,mul} cover the integer packets above but are not
152// defined for floating point, so these fold the packet by hand.
153#define EIGEN_CLANG_PACKET_REDUX_FLOAT(PACKET_TYPE) \
154 template <> \
155 EIGEN_STRONG_INLINE unpacket_traits<PACKET_TYPE>::type predux<PACKET_TYPE>(const PACKET_TYPE& a) { \
156 return detail::tree_reduce(a, [](const auto& x, const auto& y) { return x + y; }); \
157 } \
158 template <> \
159 EIGEN_STRONG_INLINE unpacket_traits<PACKET_TYPE>::type predux_mul<PACKET_TYPE>(const PACKET_TYPE& a) { \
160 return detail::tree_reduce(a, [](const auto& x, const auto& y) { return x * y; }); \
161 }
162
163EIGEN_CLANG_PACKET_REDUX_FLOAT(PacketXf)
164EIGEN_CLANG_PACKET_REDUX_FLOAT(PacketXd)
165#undef EIGEN_CLANG_PACKET_REDUX_FLOAT
166
167// --- predux_min and predux_max for float and double ---
168// Also covers the NaN-propagation variants, whose generic fallback spills the
169// packet to the stack and reduces it with a scalar loop.
170#define EIGEN_CLANG_PACKET_REDUX_MINMAX_FLOAT(PACKET_TYPE) \
171 template <> \
172 EIGEN_STRONG_INLINE unpacket_traits<PACKET_TYPE>::type predux_min(const PACKET_TYPE& a) { \
173 return detail::tree_reduce(a, detail::pmin_op()); \
174 } \
175 template <> \
176 EIGEN_STRONG_INLINE unpacket_traits<PACKET_TYPE>::type predux_max(const PACKET_TYPE& a) { \
177 return detail::tree_reduce(a, detail::pmax_op()); \
178 } \
179 template <> \
180 EIGEN_STRONG_INLINE unpacket_traits<PACKET_TYPE>::type predux_min<PropagateNumbers, PACKET_TYPE>( \
181 const PACKET_TYPE& a) { \
182 return detail::tree_reduce(a, detail::pmin_num_op()); \
183 } \
184 template <> \
185 EIGEN_STRONG_INLINE unpacket_traits<PACKET_TYPE>::type predux_max<PropagateNumbers, PACKET_TYPE>( \
186 const PACKET_TYPE& a) { \
187 return detail::tree_reduce(a, detail::pmax_num_op()); \
188 } \
189 template <> \
190 EIGEN_STRONG_INLINE unpacket_traits<PACKET_TYPE>::type predux_min<PropagateNaN, PACKET_TYPE>(const PACKET_TYPE& a) { \
191 return detail::tree_reduce(a, detail::pmin_nan_op()); \
192 } \
193 template <> \
194 EIGEN_STRONG_INLINE unpacket_traits<PACKET_TYPE>::type predux_max<PropagateNaN, PACKET_TYPE>(const PACKET_TYPE& a) { \
195 return detail::tree_reduce(a, detail::pmax_nan_op()); \
196 }
197
198EIGEN_CLANG_PACKET_REDUX_MINMAX_FLOAT(PacketXf)
199EIGEN_CLANG_PACKET_REDUX_MINMAX_FLOAT(PacketXd)
200#undef EIGEN_CLANG_PACKET_REDUX_MINMAX_FLOAT
201
202// --- predux and predux_mul for complex ---
203// The real vector of a complex packet interleaves real and imaginary parts, so
204// summing it into an (even, odd) pair accumulates the two parts separately.
205#define EIGEN_CLANG_COMPLEX_REDUX(PACKET_TYPE) \
206 template <> \
207 EIGEN_STRONG_INLINE unpacket_traits<PACKET_TYPE>::type predux<PACKET_TYPE>(const PACKET_TYPE& a) { \
208 const auto re_im = detail::reduce_add_pairs(a.v); \
209 return unpacket_traits<PACKET_TYPE>::type(re_im.first, re_im.second); \
210 } \
211 template <> \
212 EIGEN_STRONG_INLINE unpacket_traits<PACKET_TYPE>::type predux_mul<PACKET_TYPE>(const PACKET_TYPE& a) { \
213 return detail::complex_predux_mul<unpacket_traits<PACKET_TYPE>::size>::run(a); \
214 }
215
216EIGEN_CLANG_COMPLEX_REDUX(PacketXcf)
217EIGEN_CLANG_COMPLEX_REDUX(PacketXcd)
218#undef EIGEN_CLANG_COMPLEX_REDUX
219
220#endif
221
222} // end namespace internal
223} // end namespace Eigen
224
225#endif // EIGEN_REDUCTIONS_CLANG_H