Eigen  5.0.1
 
Loading...
Searching...
No Matches
AssignEvaluator.h
1// This file is part of Eigen, a lightweight C++ template library
2// for linear algebra.
3//
4// Copyright (C) 2011 Benoit Jacob <jacob.benoit.1@gmail.com>
5// Copyright (C) 2011-2014 Gael Guennebaud <gael.guennebaud@inria.fr>
6// Copyright (C) 2011-2012 Jitse Niesen <jitse@maths.leeds.ac.uk>
7//
8// This Source Code Form is subject to the terms of the Mozilla
9// Public License v. 2.0. If a copy of the MPL was not distributed
10// with this file, You can obtain one at http://mozilla.org/MPL/2.0/.
11// SPDX-License-Identifier: MPL-2.0
12
13#ifndef EIGEN_ASSIGN_EVALUATOR_H
14#define EIGEN_ASSIGN_EVALUATOR_H
15
16// IWYU pragma: private
17#include "./InternalHeaderCheck.h"
18
19namespace Eigen {
20
21// This implementation is based on Assign.h
22
23namespace internal {
24
25/***************************************************************************
26 * Part 1 : the logic deciding a strategy for traversal and unrolling *
27 ***************************************************************************/
28
29// copy_using_evaluator_traits is based on assign_traits
30
31template <typename DstEvaluator, typename SrcEvaluator, typename AssignFunc, int MaxPacketSize = Dynamic>
32struct copy_using_evaluator_traits {
33 using Src = typename SrcEvaluator::XprType;
34 using Dst = typename DstEvaluator::XprType;
35 using DstScalar = typename Dst::Scalar;
36
37 static constexpr int DstFlags = DstEvaluator::Flags;
38 static constexpr int SrcFlags = SrcEvaluator::Flags;
39
40 public:
41 static constexpr int DstAlignment = DstEvaluator::Alignment;
42 static constexpr int SrcAlignment = SrcEvaluator::Alignment;
43 static constexpr int JointAlignment = plain_enum_min(DstAlignment, SrcAlignment);
44 static constexpr bool DstHasDirectAccess = bool(DstFlags & DirectAccessBit);
45 static constexpr bool SrcIsRowMajor = bool(SrcFlags & RowMajorBit);
46 static constexpr bool DstIsRowMajor = bool(DstFlags & RowMajorBit);
47 static constexpr bool IsVectorAtCompileTime = Dst::IsVectorAtCompileTime;
48 static constexpr int RowsAtCompileTime = size_prefer_fixed(Src::RowsAtCompileTime, Dst::RowsAtCompileTime);
49 static constexpr int ColsAtCompileTime = size_prefer_fixed(Src::ColsAtCompileTime, Dst::ColsAtCompileTime);
50 static constexpr int SizeAtCompileTime = size_at_compile_time(RowsAtCompileTime, ColsAtCompileTime);
51 static constexpr int MaxRowsAtCompileTime =
52 min_size_prefer_fixed(Src::MaxRowsAtCompileTime, Dst::MaxRowsAtCompileTime);
53 static constexpr int MaxColsAtCompileTime =
54 min_size_prefer_fixed(Src::MaxColsAtCompileTime, Dst::MaxColsAtCompileTime);
55 static constexpr int MaxSizeAtCompileTime =
56 min_size_prefer_fixed(Src::MaxSizeAtCompileTime, Dst::MaxSizeAtCompileTime);
57 static constexpr int InnerSizeAtCompileTime = IsVectorAtCompileTime ? SizeAtCompileTime
58 : DstIsRowMajor ? ColsAtCompileTime
59 : RowsAtCompileTime;
60 static constexpr int MaxInnerSizeAtCompileTime = IsVectorAtCompileTime ? MaxSizeAtCompileTime
61 : DstIsRowMajor ? MaxColsAtCompileTime
62 : MaxRowsAtCompileTime;
63 static constexpr int RestrictedInnerSize = min_size_prefer_fixed(MaxInnerSizeAtCompileTime, MaxPacketSize);
64 static constexpr int RestrictedLinearSize = min_size_prefer_fixed(MaxSizeAtCompileTime, MaxPacketSize);
65 static constexpr int OuterStride = outer_stride_at_compile_time<Dst>::value;
66
67 // LinearVectorizedTraversal handles a partial-packet tail (scalar emits under
68 // complete unrolling, packet_segment under no-unrolling), so we can prefer a
69 // wider packet than find_best_packet's exact-divisor pick when that strictly
70 // reduces total op count -- e.g. N=9 float on AVX2 gets 1*Packet8f + 1 scalar
71 // instead of 2*Packet4f + 1 scalar. find_assign_linear_packet stays on the
72 // exact-divisor choice when the wider alternative is no better
73 // (e.g. N=6 double on AVX-512: 3*Packet2d ties with 1*Packet4d + 2 scalars,
74 // so Packet2d is kept and LLT/LDLT sub-vector kernels are not disturbed).
75 // InnerVectorizedTraversal still requires Size % PacketSize == 0, so its
76 // packet type continues to use find_best_packet.
77 using LinearPacketType = typename find_assign_linear_packet<DstScalar, RestrictedLinearSize>::type;
78 using InnerPacketType = typename find_best_packet<DstScalar, RestrictedInnerSize>::type;
79
80 static constexpr int LinearPacketSize = unpacket_traits<LinearPacketType>::size;
81 static constexpr int InnerPacketSize = unpacket_traits<InnerPacketType>::size;
82 // Use the exact-divisor packet size for the Inner-vs-Linear choice below, so
83 // find_assign_linear_packet's widening only changes the packet *within*
84 // LinearVectorizedTraversal and never flips an inner-vectorizable assignment
85 // onto the linear path (e.g. vectorization_logic Matrix57 on NEON int).
86 static constexpr int LinearTraversalPacketSize =
87 unpacket_traits<typename find_best_packet<DstScalar, RestrictedLinearSize>::type>::size;
88
89 public:
90 static constexpr int LinearRequiredAlignment = unpacket_traits<LinearPacketType>::alignment;
91 static constexpr int InnerRequiredAlignment = unpacket_traits<InnerPacketType>::alignment;
92
93 private:
94 static constexpr bool StorageOrdersAgree = DstIsRowMajor == SrcIsRowMajor;
95 static constexpr bool MightVectorize = StorageOrdersAgree && bool(DstFlags & SrcFlags & ActualPacketAccessBit) &&
96 bool(functor_traits<AssignFunc>::PacketAccess);
97 // Generic packet assignment stores forward from coeffRef(); the swap kernel uses writePacket().
98 static constexpr bool MayInnerVectorize =
99 MightVectorize && (DstHasDirectAccess || std::is_same<AssignFunc, swap_assign_op<DstScalar>>::value) &&
100 (InnerSizeAtCompileTime != Dynamic) && (InnerSizeAtCompileTime % InnerPacketSize == 0) &&
101 (OuterStride != Dynamic) && (OuterStride % InnerPacketSize == 0) &&
102 (EIGEN_UNALIGNED_VECTORIZE || JointAlignment >= InnerRequiredAlignment);
103 static constexpr bool MayLinearize = StorageOrdersAgree && (DstFlags & SrcFlags & LinearAccessBit);
104 static constexpr bool MayLinearVectorize =
105 MightVectorize && MayLinearize && DstHasDirectAccess &&
106 (EIGEN_UNALIGNED_VECTORIZE || (DstAlignment >= LinearRequiredAlignment) || MaxSizeAtCompileTime == Dynamic) &&
107 (MaxSizeAtCompileTime == Dynamic || MaxSizeAtCompileTime >= LinearPacketSize);
108 /* If the destination isn't aligned, we have to do runtime checks and we don't unroll,
109 so it's only good for large enough sizes. */
110 static constexpr int InnerSizeThreshold = (EIGEN_UNALIGNED_VECTORIZE ? 1 : 3) * InnerPacketSize;
111 static constexpr bool MaySliceVectorize =
112 MightVectorize && DstHasDirectAccess &&
113 (MaxInnerSizeAtCompileTime == Dynamic || MaxInnerSizeAtCompileTime >= InnerSizeThreshold);
114 /* slice vectorization can be slow, so we only want it if the slices are big, which is
115 indicated by InnerMaxSize rather than InnerSize, think of the case of a dynamic block
116 in a fixed-size matrix
117 However, with EIGEN_UNALIGNED_VECTORIZE and unrolling, slice vectorization is still worth it */
118
119 public:
120 static constexpr int Traversal = SizeAtCompileTime == 0 ? AllAtOnceTraversal
121 : (MayLinearVectorize && (LinearTraversalPacketSize > InnerPacketSize))
122 ? LinearVectorizedTraversal
123 : MayInnerVectorize ? InnerVectorizedTraversal
124 : MayLinearVectorize ? LinearVectorizedTraversal
125 : MaySliceVectorize ? SliceVectorizedTraversal
126 : MayLinearize ? LinearTraversal
127 : DefaultTraversal;
128 static constexpr bool Vectorized = Traversal == InnerVectorizedTraversal || Traversal == LinearVectorizedTraversal ||
129 Traversal == SliceVectorizedTraversal;
130
131 using PacketType = std::conditional_t<Traversal == LinearVectorizedTraversal, LinearPacketType, InnerPacketType>;
132
133 private:
134 static constexpr int ActualPacketSize = Vectorized ? unpacket_traits<PacketType>::size : 1;
135 static constexpr int UnrollingLimit = EIGEN_UNROLLING_LIMIT * ActualPacketSize;
136 static constexpr int CoeffReadCost = int(DstEvaluator::CoeffReadCost) + int(SrcEvaluator::CoeffReadCost);
137 static constexpr bool MayUnrollInner =
138 (InnerSizeAtCompileTime != Dynamic) && (InnerSizeAtCompileTime * CoeffReadCost <= UnrollingLimit);
139
140 public:
141 // True when the whole assignment is a fixed-size kernel cheap enough to emit as straight-line
142 // code. Selects CompleteUnrolling, and gates the scalar tail in the SliceVectorized loop below.
143 static constexpr bool MayUnrollCompletely =
144 (SizeAtCompileTime != Dynamic) && (SizeAtCompileTime * CoeffReadCost <= UnrollingLimit);
145 static constexpr int Unrolling =
146 (Traversal == InnerVectorizedTraversal || Traversal == DefaultTraversal)
147 ? (MayUnrollCompletely ? CompleteUnrolling
148 : MayUnrollInner ? InnerUnrolling
149 : NoUnrolling)
150 : Traversal == LinearVectorizedTraversal
151 ? (MayUnrollCompletely && (EIGEN_UNALIGNED_VECTORIZE || (DstAlignment >= LinearRequiredAlignment))
152 ? CompleteUnrolling
153 : NoUnrolling)
154 : Traversal == LinearTraversal ? (MayUnrollCompletely ? CompleteUnrolling : NoUnrolling)
155#if EIGEN_UNALIGNED_VECTORIZE
156 : Traversal == SliceVectorizedTraversal ? (MayUnrollInner ? InnerUnrolling : NoUnrolling)
157#endif
158 : NoUnrolling;
159 // Scalar tails for expressions bounded to <= 4 packets (runtime-sized blocks of small fixed matrices, as in the
160 // column steps of a 4x4 Cholesky): their results are reread almost at once, and a masked store does not forward to
161 // the following loads. That pays only while a tail is a few cheap coefficients: packets of <= 8 lanes, and a source
162 // cost below HugeCost (a lazy product with a runtime inner size computes a dot product per coefficient).
163 static constexpr bool UsePacketSegment =
164 has_packet_segment<PacketType>::value &&
165 !(MaxSizeAtCompileTime != Dynamic && MaxSizeAtCompileTime <= 4 * int(unpacket_traits<PacketType>::size) &&
166 unpacket_traits<PacketType>::size <= 8 && int(SrcEvaluator::CoeffReadCost) < HugeCost);
167
168#ifdef EIGEN_DEBUG_ASSIGN
169 static void debug() {
170 std::cerr << "DstXpr: " << typeid(typename DstEvaluator::XprType).name() << std::endl;
171 std::cerr << "SrcXpr: " << typeid(typename SrcEvaluator::XprType).name() << std::endl;
172 std::cerr.setf(std::ios::hex, std::ios::basefield);
173 std::cerr << "DstFlags"
174 << " = " << DstFlags << " (" << demangle_flags(DstFlags) << " )" << std::endl;
175 std::cerr << "SrcFlags"
176 << " = " << SrcFlags << " (" << demangle_flags(SrcFlags) << " )" << std::endl;
177 std::cerr.unsetf(std::ios::hex);
178 EIGEN_DEBUG_VAR(DstAlignment)
179 EIGEN_DEBUG_VAR(SrcAlignment)
180 EIGEN_DEBUG_VAR(LinearRequiredAlignment)
181 EIGEN_DEBUG_VAR(InnerRequiredAlignment)
182 EIGEN_DEBUG_VAR(JointAlignment)
183 EIGEN_DEBUG_VAR(InnerSizeAtCompileTime)
184 EIGEN_DEBUG_VAR(MaxInnerSizeAtCompileTime)
185 EIGEN_DEBUG_VAR(LinearPacketSize)
186 EIGEN_DEBUG_VAR(InnerPacketSize)
187 EIGEN_DEBUG_VAR(ActualPacketSize)
188 EIGEN_DEBUG_VAR(StorageOrdersAgree)
189 EIGEN_DEBUG_VAR(MightVectorize)
190 EIGEN_DEBUG_VAR(MayLinearize)
191 EIGEN_DEBUG_VAR(MayInnerVectorize)
192 EIGEN_DEBUG_VAR(MayLinearVectorize)
193 EIGEN_DEBUG_VAR(MaySliceVectorize)
194 std::cerr << "Traversal"
195 << " = " << Traversal << " (" << demangle_traversal(Traversal) << ")" << std::endl;
196 EIGEN_DEBUG_VAR(SrcEvaluator::CoeffReadCost)
197 EIGEN_DEBUG_VAR(DstEvaluator::CoeffReadCost)
198 EIGEN_DEBUG_VAR(Dst::SizeAtCompileTime)
199 EIGEN_DEBUG_VAR(UnrollingLimit)
200 EIGEN_DEBUG_VAR(MayUnrollCompletely)
201 EIGEN_DEBUG_VAR(MayUnrollInner)
202 std::cerr << "Unrolling"
203 << " = " << Unrolling << " (" << demangle_unrolling(Unrolling) << ")" << std::endl;
204 std::cerr << std::endl;
205 }
206#endif
207};
208
209/***************************************************************************
210 * Part 2 : meta-unrollers
211 ***************************************************************************/
212
213/************************
214*** Default traversal ***
215************************/
216
217template <typename Kernel, int Index_, int Stop>
218struct copy_using_evaluator_DefaultTraversal_CompleteUnrolling {
219 template <int... Offsets>
220 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run_impl(Kernel& kernel,
221 std::integer_sequence<int, Offsets...>) {
222 int unused[] = {
223 0, (kernel.assignCoeffByOuterInner((Index_ + Offsets) / Kernel::AssignmentTraits::InnerSizeAtCompileTime,
224 (Index_ + Offsets) % Kernel::AssignmentTraits::InnerSizeAtCompileTime),
225 0)...};
226 EIGEN_UNUSED_VARIABLE(unused);
227 }
228
229 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run(Kernel& kernel) {
230 run_impl(kernel, std::make_integer_sequence<int, Stop - Index_>{});
231 }
232};
233
234template <typename Kernel, int Index_, int Stop>
235struct copy_using_evaluator_DefaultTraversal_InnerUnrolling {
236 template <int... Offsets>
237 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run_impl(Kernel& kernel, Index outer,
238 std::integer_sequence<int, Offsets...>) {
239 int unused[] = {0, (kernel.assignCoeffByOuterInner(outer, Index_ + Offsets), 0)...};
240 EIGEN_UNUSED_VARIABLE(unused);
241 }
242
243 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run(Kernel& kernel, Index outer) {
244 run_impl(kernel, outer, std::make_integer_sequence<int, Stop - Index_>{});
245 }
246};
247
248/***********************
249*** Linear traversal ***
250***********************/
251
252template <typename Kernel, int Index_, int Stop>
253struct copy_using_evaluator_LinearTraversal_CompleteUnrolling {
254 template <int... Offsets>
255 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run_impl(Kernel& kernel,
256 std::integer_sequence<int, Offsets...>) {
257 int unused[] = {0, (kernel.assignCoeff(Index_ + Offsets), 0)...};
258 EIGEN_UNUSED_VARIABLE(unused);
259 }
260
261 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run(Kernel& kernel) {
262 run_impl(kernel, std::make_integer_sequence<int, Stop - Index_>{});
263 }
264};
265
266/**************************
267*** Inner vectorization ***
268**************************/
269
270template <typename Kernel, int Index_, int Stop>
271struct copy_using_evaluator_innervec_CompleteUnrolling {
272 using PacketType = typename Kernel::PacketType;
273 static constexpr int PacketSize = unpacket_traits<PacketType>::size;
274 static constexpr int SrcAlignment = Kernel::AssignmentTraits::SrcAlignment;
275 static constexpr int DstAlignment = Kernel::AssignmentTraits::DstAlignment;
276 static_assert((Stop - Index_) % PacketSize == 0, "Packet unrolling range must be divisible by packet size.");
277
278 template <int... Offsets>
279 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run_impl(Kernel& kernel, std::integer_sequence<int, Offsets...>) {
280 int unused[] = {0, (kernel.template assignPacketByOuterInner<DstAlignment, SrcAlignment, PacketType>(
281 (Index_ + Offsets * PacketSize) / Kernel::AssignmentTraits::InnerSizeAtCompileTime,
282 (Index_ + Offsets * PacketSize) % Kernel::AssignmentTraits::InnerSizeAtCompileTime),
283 0)...};
284 EIGEN_UNUSED_VARIABLE(unused);
285 }
286
287 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel& kernel) {
288 run_impl(kernel, std::make_integer_sequence<int, (Stop - Index_) / PacketSize>{});
289 }
290};
291
292template <typename Kernel, int Index_, int Stop, int SrcAlignment, int DstAlignment>
293struct copy_using_evaluator_innervec_InnerUnrolling {
294 using PacketType = typename Kernel::PacketType;
295 static constexpr int PacketSize = unpacket_traits<PacketType>::size;
296 static_assert((Stop - Index_) % PacketSize == 0, "Packet unrolling range must be divisible by packet size.");
297
298 template <int... Offsets>
299 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run_impl(Kernel& kernel, Index outer,
300 std::integer_sequence<int, Offsets...>) {
301 int unused[] = {0, (kernel.template assignPacketByOuterInner<DstAlignment, SrcAlignment, PacketType>(
302 outer, Index_ + Offsets * PacketSize),
303 0)...};
304 EIGEN_UNUSED_VARIABLE(unused);
305 EIGEN_UNUSED_VARIABLE(outer);
306 }
307
308 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel& kernel, Index outer) {
309 run_impl(kernel, outer, std::make_integer_sequence<int, (Stop - Index_) / PacketSize>{});
310 }
311};
312
313template <typename Kernel, int Start, int Stop, int SrcAlignment, int DstAlignment, bool UsePacketSegment>
314struct copy_using_evaluator_innervec_segment {
315 using PacketType = typename Kernel::PacketType;
316
317 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel& kernel, Index outer) {
318 kernel.template assignPacketSegmentByOuterInner<DstAlignment, SrcAlignment, PacketType>(outer, Start, 0,
319 Stop - Start);
320 }
321};
322
323template <typename Kernel, int Start, int Stop, int SrcAlignment, int DstAlignment>
324struct copy_using_evaluator_innervec_segment<Kernel, Start, Stop, SrcAlignment, DstAlignment,
325 /*UsePacketSegment*/ false>
326 : copy_using_evaluator_DefaultTraversal_InnerUnrolling<Kernel, Start, Stop> {};
327
328template <typename Kernel, int Stop, int SrcAlignment, int DstAlignment>
329struct copy_using_evaluator_innervec_segment<Kernel, Stop, Stop, SrcAlignment, DstAlignment,
330 /*UsePacketSegment*/ true> {
331 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run(Kernel&, Index) {}
332};
333
334template <typename Kernel, int Stop, int SrcAlignment, int DstAlignment>
335struct copy_using_evaluator_innervec_segment<Kernel, Stop, Stop, SrcAlignment, DstAlignment,
336 /*UsePacketSegment*/ false> {
337 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run(Kernel&, Index) {}
338};
339
340/***************************************************************************
341 * Part 3 : implementation of all cases
342 ***************************************************************************/
343
344// dense_assignment_loop is based on assign_impl
345
346template <typename Kernel, int Traversal = Kernel::AssignmentTraits::Traversal,
347 int Unrolling = Kernel::AssignmentTraits::Unrolling>
348struct dense_assignment_loop_impl;
349
350template <typename Kernel, int Traversal = Kernel::AssignmentTraits::Traversal,
351 int Unrolling = Kernel::AssignmentTraits::Unrolling>
352struct dense_assignment_loop {
353 // Always inlined so the loop lands in the caller that owns the evaluators (see the slice traversal).
354 EIGEN_DEVICE_FUNC static EIGEN_ALWAYS_INLINE constexpr void run(Kernel& kernel) {
355#ifdef __cpp_lib_is_constant_evaluated
356 if (internal::is_constant_evaluated())
357 dense_assignment_loop_impl<Kernel, Traversal == AllAtOnceTraversal ? AllAtOnceTraversal : DefaultTraversal,
358 NoUnrolling>::run(kernel);
359 else
360#endif
361 dense_assignment_loop_impl<Kernel, Traversal, Unrolling>::run(kernel);
362 }
363};
364
365/************************
366***** Special Cases *****
367************************/
368
369// Zero-sized assignment is a no-op.
370template <typename Kernel, int Unrolling>
371struct dense_assignment_loop_impl<Kernel, AllAtOnceTraversal, Unrolling> {
372 static constexpr int SizeAtCompileTime = Kernel::AssignmentTraits::SizeAtCompileTime;
373
374 EIGEN_DEVICE_FUNC static void EIGEN_STRONG_INLINE constexpr run(Kernel& /*kernel*/) {
375 EIGEN_STATIC_ASSERT(SizeAtCompileTime == 0, EIGEN_INTERNAL_ERROR_PLEASE_FILE_A_BUG_REPORT)
376 }
377};
378
379/************************
380*** Default traversal ***
381************************/
382
383template <typename Kernel>
384struct dense_assignment_loop_impl<Kernel, DefaultTraversal, NoUnrolling> {
385 EIGEN_DEVICE_FUNC static void EIGEN_STRONG_INLINE constexpr run(Kernel& kernel) {
386 for (Index outer = 0; outer < kernel.outerSize(); ++outer) {
387 for (Index inner = 0; inner < kernel.innerSize(); ++inner) {
388 kernel.assignCoeffByOuterInner(outer, inner);
389 }
390 }
391 }
392};
393
394template <typename Kernel>
395struct dense_assignment_loop_impl<Kernel, DefaultTraversal, CompleteUnrolling> {
396 static constexpr int SizeAtCompileTime = Kernel::AssignmentTraits::SizeAtCompileTime;
397
398 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run(Kernel& kernel) {
399 copy_using_evaluator_DefaultTraversal_CompleteUnrolling<Kernel, 0, SizeAtCompileTime>::run(kernel);
400 }
401};
402
403template <typename Kernel>
404struct dense_assignment_loop_impl<Kernel, DefaultTraversal, InnerUnrolling> {
405 static constexpr int InnerSizeAtCompileTime = Kernel::AssignmentTraits::InnerSizeAtCompileTime;
406
407 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run(Kernel& kernel) {
408 const Index outerSize = kernel.outerSize();
409 for (Index outer = 0; outer < outerSize; ++outer)
410 copy_using_evaluator_DefaultTraversal_InnerUnrolling<Kernel, 0, InnerSizeAtCompileTime>::run(kernel, outer);
411 }
412};
413
414/***************************
415*** Linear vectorization ***
416***************************/
417
418// The goal of unaligned_dense_assignment_loop is simply to factorize the handling
419// of the non vectorizable beginning and ending parts
420
421template <typename PacketType, int DstAlignment, int SrcAlignment, bool UsePacketSegment, bool Skip>
422struct unaligned_dense_assignment_loop {
423 // if Skip == true, then do nothing
424 template <typename Kernel>
425 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run(Kernel& /*kernel*/, Index /*start*/, Index /*end*/) {}
426 template <typename Kernel>
427 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run(Kernel& /*kernel*/, Index /*outer*/,
428 Index /*innerStart*/, Index /*innerEnd*/) {}
429};
430
431template <typename PacketType, int DstAlignment, int SrcAlignment>
432struct unaligned_dense_assignment_loop<PacketType, DstAlignment, SrcAlignment, /*UsePacketSegment*/ true,
433 /*Skip*/ false> {
434 template <typename Kernel>
435 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run(Kernel& kernel, Index start, Index end) {
436 Index count = end - start;
437 eigen_assert(count <= unpacket_traits<PacketType>::size);
438 if (count > 0) kernel.template assignPacketSegment<DstAlignment, SrcAlignment, PacketType>(start, 0, count);
439 }
440 template <typename Kernel>
441 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run(Kernel& kernel, Index outer, Index start, Index end) {
442 Index count = end - start;
443 eigen_assert(count <= unpacket_traits<PacketType>::size);
444 if (count > 0)
445 kernel.template assignPacketSegmentByOuterInner<DstAlignment, SrcAlignment, PacketType>(outer, start, 0, count);
446 }
447};
448
449template <typename PacketType, int DstAlignment, int SrcAlignment>
450struct unaligned_dense_assignment_loop<PacketType, DstAlignment, SrcAlignment, /*UsePacketSegment*/ false,
451 /*Skip*/ false> {
452 template <typename Kernel>
453 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run(Kernel& kernel, Index start, Index end) {
454 for (Index index = start; index < end; ++index) kernel.assignCoeff(index);
455 }
456 template <typename Kernel>
457 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run(Kernel& kernel, Index outer, Index innerStart,
458 Index innerEnd) {
459 for (Index inner = innerStart; inner < innerEnd; ++inner) kernel.assignCoeffByOuterInner(outer, inner);
460 }
461};
462
463template <typename Kernel, int Index_, int Stop>
464struct copy_using_evaluator_linearvec_CompleteUnrolling {
465 using PacketType = typename Kernel::PacketType;
466 static constexpr int PacketSize = unpacket_traits<PacketType>::size;
467 static constexpr int SrcAlignment = Kernel::AssignmentTraits::SrcAlignment;
468 static constexpr int DstAlignment = Kernel::AssignmentTraits::DstAlignment;
469 static_assert((Stop - Index_) % PacketSize == 0, "Packet unrolling range must be divisible by packet size.");
470
471 template <int... Offsets>
472 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run_impl(Kernel& kernel, std::integer_sequence<int, Offsets...>) {
473 int unused[] = {
474 0, (kernel.template assignPacket<DstAlignment, SrcAlignment, PacketType>(Index_ + Offsets * PacketSize), 0)...};
475 EIGEN_UNUSED_VARIABLE(unused);
476 }
477
478 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel& kernel) {
479 run_impl(kernel, std::make_integer_sequence<int, (Stop - Index_) / PacketSize>{});
480 }
481};
482
483template <typename Kernel, int Index_, int Stop, bool UsePacketSegment>
484struct copy_using_evaluator_linearvec_segment {
485 using PacketType = typename Kernel::PacketType;
486 static constexpr int SrcAlignment = Kernel::AssignmentTraits::SrcAlignment;
487 static constexpr int DstAlignment = Kernel::AssignmentTraits::DstAlignment;
488
489 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel& kernel) {
490 kernel.template assignPacketSegment<DstAlignment, SrcAlignment, PacketType>(Index_, 0, Stop - Index_);
491 }
492};
493
494template <typename Kernel, int Index_, int Stop>
495struct copy_using_evaluator_linearvec_segment<Kernel, Index_, Stop, /*UsePacketSegment*/ false>
496 : copy_using_evaluator_LinearTraversal_CompleteUnrolling<Kernel, Index_, Stop> {};
497
498template <typename Kernel, int Stop>
499struct copy_using_evaluator_linearvec_segment<Kernel, Stop, Stop, /*UsePacketSegment*/ true> {
500 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run(Kernel&) {}
501};
502
503template <typename Kernel, int Stop>
504struct copy_using_evaluator_linearvec_segment<Kernel, Stop, Stop, /*UsePacketSegment*/ false> {
505 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run(Kernel&) {}
506};
507
508template <typename Kernel>
509struct dense_assignment_loop_impl<Kernel, LinearVectorizedTraversal, NoUnrolling> {
510 using Scalar = typename Kernel::Scalar;
511 using PacketType = typename Kernel::PacketType;
512 static constexpr int PacketSize = unpacket_traits<PacketType>::size;
513 static constexpr int SrcAlignment = Kernel::AssignmentTraits::JointAlignment;
514 static constexpr int DstAlignment = plain_enum_max(Kernel::AssignmentTraits::DstAlignment, alignof(Scalar));
515 static constexpr int RequestedAlignment = unpacket_traits<PacketType>::alignment;
516 static constexpr bool Alignable = (DstAlignment >= RequestedAlignment) ||
517 (static_cast<std::size_t>(RequestedAlignment - DstAlignment) % sizeof(Scalar) == 0);
518 static constexpr int Alignment = Alignable ? RequestedAlignment : DstAlignment;
519 static constexpr bool DstIsAligned = DstAlignment >= Alignment;
520 static constexpr bool UsePacketSegment = Kernel::AssignmentTraits::UsePacketSegment;
521
522 using head_loop =
523 unaligned_dense_assignment_loop<PacketType, DstAlignment, SrcAlignment, UsePacketSegment, DstIsAligned>;
524 using tail_loop = unaligned_dense_assignment_loop<PacketType, Alignment, SrcAlignment, UsePacketSegment, false>;
525
526 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run(Kernel& kernel) {
527 // When a small bound disables available packet segments, GCC's -Waggressive-loop-optimizations fires on the
528 // scalar head and tail loops unless they see that bound; the no-op clamp supplies it. Backends without segments
529 // keep the unclamped size and their existing codegen.
530 constexpr int MaxSize = Kernel::AssignmentTraits::MaxSizeAtCompileTime;
531 constexpr bool ClampToMaxSize = has_packet_segment<PacketType>::value && !UsePacketSegment && MaxSize != Dynamic;
532 eigen_assert(MaxSize == Dynamic || kernel.size() <= MaxSize);
533 const Index size = ClampToMaxSize ? numext::mini(kernel.size(), Index(MaxSize)) : kernel.size();
534 const Index alignedStart = DstIsAligned ? 0 : first_aligned<Alignment>(kernel.dstDataPtr(), size);
535
536 head_loop::run(kernel, 0, alignedStart);
537
538 for (Index index = alignedStart; index <= size - PacketSize; index += PacketSize)
539 kernel.template assignPacket<Alignment, SrcAlignment, PacketType>(index);
540
541 // Derive the tail bound directly; GCC can lose its range when it comes from the packet-loop induction variable.
542 const Index alignedEnd = size - (size - alignedStart) % PacketSize;
543 tail_loop::run(kernel, alignedEnd, size);
544 }
545};
546
547template <typename Kernel>
548struct dense_assignment_loop_impl<Kernel, LinearVectorizedTraversal, CompleteUnrolling> {
549 using PacketType = typename Kernel::PacketType;
550 static constexpr int PacketSize = unpacket_traits<PacketType>::size;
551 static constexpr int Size = Kernel::AssignmentTraits::SizeAtCompileTime;
552 static constexpr int AlignedSize = numext::round_down(Size, PacketSize);
553
554 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run(Kernel& kernel) {
555 copy_using_evaluator_linearvec_CompleteUnrolling<Kernel, 0, AlignedSize>::run(kernel);
556 // Partial-packet tail. Unlike the NoUnrolling case above, here the size is a
557 // compile-time constant and the loop is fully unrolled, so the tail is a
558 // fixed handful (fewer than PacketSize) of scalar assignments. Emit those
559 // directly by forcing UsePacketSegment = false: a masked packet segment
560 // buys nothing when there is no runtime-variable trip count to collapse,
561 // and on the AVX backend it lowers to a masked store (vmaskmovps/vmaskmovpd)
562 // that is slow on current hardware and forwards poorly to later loads. The
563 // NoUnrolling loop keeps UsePacketSegment because there the tail length is a
564 // runtime value, where a single masked op does beat a scalar remainder loop.
565 copy_using_evaluator_linearvec_segment<Kernel, AlignedSize, Size, /*UsePacketSegment=*/false>::run(kernel);
566 }
567};
568
569/**************************
570*** Inner vectorization ***
571**************************/
572
573template <typename Kernel>
574struct dense_assignment_loop_impl<Kernel, InnerVectorizedTraversal, NoUnrolling> {
575 using PacketType = typename Kernel::PacketType;
576 static constexpr int PacketSize = unpacket_traits<PacketType>::size;
577 static constexpr int SrcAlignment = Kernel::AssignmentTraits::JointAlignment;
578 static constexpr int DstAlignment = Kernel::AssignmentTraits::DstAlignment;
579
580 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run(Kernel& kernel) {
581 const Index innerSize = kernel.innerSize();
582 const Index outerSize = kernel.outerSize();
583 for (Index outer = 0; outer < outerSize; ++outer)
584 for (Index inner = 0; inner < innerSize; inner += PacketSize)
585 kernel.template assignPacketByOuterInner<DstAlignment, SrcAlignment, PacketType>(outer, inner);
586 }
587};
588
589template <typename Kernel>
590struct dense_assignment_loop_impl<Kernel, InnerVectorizedTraversal, CompleteUnrolling> {
591 static constexpr int SizeAtCompileTime = Kernel::AssignmentTraits::SizeAtCompileTime;
592
593 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel& kernel) {
594 copy_using_evaluator_innervec_CompleteUnrolling<Kernel, 0, SizeAtCompileTime>::run(kernel);
595 }
596};
597
598template <typename Kernel>
599struct dense_assignment_loop_impl<Kernel, InnerVectorizedTraversal, InnerUnrolling> {
600 static constexpr int InnerSize = Kernel::AssignmentTraits::InnerSizeAtCompileTime;
601 static constexpr int SrcAlignment = Kernel::AssignmentTraits::SrcAlignment;
602 static constexpr int DstAlignment = Kernel::AssignmentTraits::DstAlignment;
603
604 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(Kernel& kernel) {
605 const Index outerSize = kernel.outerSize();
606 for (Index outer = 0; outer < outerSize; ++outer)
607 copy_using_evaluator_innervec_InnerUnrolling<Kernel, 0, InnerSize, SrcAlignment, DstAlignment>::run(kernel,
608 outer);
609 }
610};
611
612/***********************
613*** Linear traversal ***
614***********************/
615
616template <typename Kernel>
617struct dense_assignment_loop_impl<Kernel, LinearTraversal, NoUnrolling> {
618 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run(Kernel& kernel) {
619 const Index size = kernel.size();
620 for (Index i = 0; i < size; ++i) kernel.assignCoeff(i);
621 }
622};
623
624template <typename Kernel>
625struct dense_assignment_loop_impl<Kernel, LinearTraversal, CompleteUnrolling> {
626 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run(Kernel& kernel) {
627 copy_using_evaluator_LinearTraversal_CompleteUnrolling<Kernel, 0, Kernel::AssignmentTraits::SizeAtCompileTime>::run(
628 kernel);
629 }
630};
631
632/**************************
633*** Slice vectorization ***
634***************************/
635
636template <typename Kernel>
637struct dense_assignment_loop_impl<Kernel, SliceVectorizedTraversal, NoUnrolling> {
638 using Scalar = typename Kernel::Scalar;
639 using PacketType = typename Kernel::PacketType;
640 static constexpr int PacketSize = unpacket_traits<PacketType>::size;
641 static constexpr int SrcAlignment = Kernel::AssignmentTraits::JointAlignment;
642 static constexpr int DstAlignment = plain_enum_max(Kernel::AssignmentTraits::DstAlignment, alignof(Scalar));
643 static constexpr int RequestedAlignment = unpacket_traits<PacketType>::alignment;
644 static constexpr bool Alignable = (DstAlignment >= RequestedAlignment) ||
645 (static_cast<std::size_t>(RequestedAlignment - DstAlignment) % sizeof(Scalar) == 0);
646 static constexpr int Alignment = Alignable ? RequestedAlignment : DstAlignment;
647 static constexpr bool DstIsAligned = DstAlignment >= Alignment;
648 static constexpr bool UsePacketSegment = Kernel::AssignmentTraits::UsePacketSegment;
649
650 using head_loop = unaligned_dense_assignment_loop<PacketType, DstAlignment, Unaligned, UsePacketSegment, !Alignable>;
651 using tail_loop = unaligned_dense_assignment_loop<PacketType, Alignment, Unaligned, UsePacketSegment, false>;
652
653 // All inner slices share one alignment offset: the loop bounds are outer-invariant and stay
654 // hoisted out of the outer loop. Keeping that loop free of per-slice bookkeeping is what makes
655 // slice vectorization competitive with compiler-vectorized scalar code when slices are short.
656 EIGEN_DEVICE_FUNC static EIGEN_ALWAYS_INLINE constexpr void runInvariant(Kernel& kernel, Index alignedStart,
657 Index innerSize, Index outerSize) {
658 const Index alignedEnd = alignedStart + numext::round_down(innerSize - alignedStart, PacketSize);
659 for (Index outer = 0; outer < outerSize; ++outer) {
660 head_loop::run(kernel, outer, 0, alignedStart);
661
662 // do the vectorizable part of the assignment
663 for (Index inner = alignedStart; inner < alignedEnd; inner += PacketSize)
664 kernel.template assignPacketByOuterInner<Alignment, Unaligned, PacketType>(outer, inner);
665
666 tail_loop::run(kernel, outer, alignedEnd, innerSize);
667 }
668 }
669
670#if EIGEN_UNALIGNED_VECTORIZE
671 // Unaligned stores with outer-invariant bounds avoid chasing each slice's alignment offset.
672 // Also used for statically aligned destinations, so run() inlines one loop nest instead of two.
673 using unaligned_tail_loop =
674 unaligned_dense_assignment_loop<PacketType, Unaligned, Unaligned, UsePacketSegment, false>;
675
676 EIGEN_DEVICE_FUNC static EIGEN_ALWAYS_INLINE constexpr void runUnaligned(Kernel& kernel, Index innerSize,
677 Index outerSize) {
678 const Index packetEnd = numext::round_down(innerSize, PacketSize);
679 for (Index outer = 0; outer < outerSize; ++outer) {
680 for (Index inner = 0; inner < packetEnd; inner += PacketSize)
681 kernel.template assignPacketByOuterInner<Unaligned, Unaligned, PacketType>(outer, inner);
682
683 unaligned_tail_loop::run(kernel, outer, packetEnd, innerSize);
684 }
685 }
686#endif
687
688 // run(), runInvariant() and runUnaligned() must inline into the function that owns the evaluators.
689 // Out of line, `kernel` is a reference the packet stores may alias, so the evaluators' pointers
690 // and strides are reloaded after every store. Neither GCC nor Clang inlines them reliably unforced.
691 EIGEN_DEVICE_FUNC static EIGEN_ALWAYS_INLINE constexpr void run(Kernel& kernel) {
692 const Index innerSize = kernel.innerSize();
693 const Index outerSize = kernel.outerSize();
694#if EIGEN_UNALIGNED_VECTORIZE
695 EIGEN_IF_CONSTEXPR (DstIsAligned) {
696 runUnaligned(kernel, innerSize, outerSize);
697 return;
698 }
699#endif
700 const Scalar* dst_ptr = kernel.dstDataPtr();
701 const Index alignedStep = Alignable ? (PacketSize - kernel.outerStride() % PacketSize) % PacketSize : 0;
702 Index alignedStart = ((!Alignable) || DstIsAligned) ? 0 : internal::first_aligned<Alignment>(dst_ptr, innerSize);
703
704 if (alignedStep == 0) {
705 runInvariant(kernel, alignedStart, innerSize, outerSize);
706 return;
707 }
708#if EIGEN_UNALIGNED_VECTORIZE
709 runUnaligned(kernel, innerSize, outerSize);
710#else
711 for (Index outer = 0; outer < outerSize; ++outer) {
712 const Index alignedEnd = alignedStart + numext::round_down(innerSize - alignedStart, PacketSize);
713
714 head_loop::run(kernel, outer, 0, alignedStart);
715
716 // do the vectorizable part of the assignment
717 for (Index inner = alignedStart; inner < alignedEnd; inner += PacketSize)
718 kernel.template assignPacketByOuterInner<Alignment, Unaligned, PacketType>(outer, inner);
719
720 tail_loop::run(kernel, outer, alignedEnd, innerSize);
721
722 alignedStart = numext::mini((alignedStart + alignedStep) % PacketSize, innerSize);
723 }
724#endif
725 }
726};
727
728#if EIGEN_UNALIGNED_VECTORIZE
729template <typename Kernel>
730struct dense_assignment_loop_impl<Kernel, SliceVectorizedTraversal, InnerUnrolling> {
731 using PacketType = typename Kernel::PacketType;
732 static constexpr int PacketSize = unpacket_traits<PacketType>::size;
733 static constexpr int InnerSize = Kernel::AssignmentTraits::InnerSizeAtCompileTime;
734 static constexpr int VectorizableSize = numext::round_down(InnerSize, PacketSize);
735
736 // The tail length is a compile-time constant here, so it can be emitted as a masked packet
737 // segment or as scalars. Scalars win for the same fixed-size kernels LinearVectorizedTraversal
738 // already emits them for: their destination is a temporary the enclosing expression reloads by
739 // packet, and on AVX a masked store forwards poorly to those loads. Everything else keeps the
740 // segment, where one packet evaluation replaces up to PacketSize - 1 scalar ones.
741 static constexpr bool UsePacketSegment =
742 Kernel::AssignmentTraits::UsePacketSegment && !Kernel::AssignmentTraits::MayUnrollCompletely;
743
744 using packet_loop = copy_using_evaluator_innervec_InnerUnrolling<Kernel, 0, VectorizableSize, Unaligned, Unaligned>;
745 using packet_segment_loop = copy_using_evaluator_innervec_segment<Kernel, VectorizableSize, InnerSize, Unaligned,
746 Unaligned, UsePacketSegment>;
747
748 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run(Kernel& kernel) {
749 for (Index outer = 0; outer < kernel.outerSize(); ++outer) {
750 packet_loop::run(kernel, outer);
751 packet_segment_loop::run(kernel, outer);
752 }
753 }
754};
755#endif
756
757/***************************************************************************
758 * Part 4 : Generic dense assignment kernel
759 ***************************************************************************/
760
761// This class generalize the assignment of a coefficient (or packet) from one dense evaluator
762// to another dense writable evaluator.
763// It is parametrized by the two evaluators, and the actual assignment functor.
764// This abstraction level permits to keep the evaluation loops as simple and as generic as possible.
765// One can customize the assignment using this generic dense_assignment_kernel with different
766// functors, or by completely overloading it, by-passing a functor.
767template <typename DstEvaluatorTypeT, typename SrcEvaluatorTypeT, typename Functor, int Version = Specialized>
768class generic_dense_assignment_kernel {
769 protected:
770 using DstXprType = typename DstEvaluatorTypeT::XprType;
771 using SrcXprType = typename SrcEvaluatorTypeT::XprType;
772
773 public:
774 using DstEvaluatorType = DstEvaluatorTypeT;
775 using SrcEvaluatorType = SrcEvaluatorTypeT;
776 using Scalar = typename DstEvaluatorType::Scalar;
777 using AssignmentTraits = copy_using_evaluator_traits<DstEvaluatorTypeT, SrcEvaluatorTypeT, Functor>;
778 using PacketType = typename AssignmentTraits::PacketType;
779
780 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE constexpr generic_dense_assignment_kernel(DstEvaluatorType& dst,
781 const SrcEvaluatorType& src,
782 const Functor& func,
783 DstXprType& dstExpr)
784 : m_dst(dst), m_src(src), m_functor(func), m_dstExpr(dstExpr) {
785#ifdef EIGEN_DEBUG_ASSIGN
786 AssignmentTraits::debug();
787#endif
788 }
789
790 EIGEN_DEVICE_FUNC constexpr Index size() const noexcept { return m_dstExpr.size(); }
791 EIGEN_DEVICE_FUNC constexpr Index innerSize() const noexcept { return m_dstExpr.innerSize(); }
792 EIGEN_DEVICE_FUNC constexpr Index outerSize() const noexcept { return m_dstExpr.outerSize(); }
793 EIGEN_DEVICE_FUNC constexpr Index rows() const noexcept { return m_dstExpr.rows(); }
794 EIGEN_DEVICE_FUNC constexpr Index cols() const noexcept { return m_dstExpr.cols(); }
795 EIGEN_DEVICE_FUNC constexpr Index outerStride() const noexcept { return m_dstExpr.outerStride(); }
796
797 EIGEN_DEVICE_FUNC constexpr DstEvaluatorType& dstEvaluator() noexcept { return m_dst; }
798 EIGEN_DEVICE_FUNC constexpr const SrcEvaluatorType& srcEvaluator() const noexcept { return m_src; }
799
801 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE constexpr void assignCoeff(Index row, Index col) {
802 m_functor.assignCoeff(m_dst.coeffRef(row, col), m_src.coeff(row, col));
803 }
804
806 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE constexpr void assignCoeff(Index index) {
807 m_functor.assignCoeff(m_dst.coeffRef(index), m_src.coeff(index));
808 }
809
811 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE constexpr void assignCoeffByOuterInner(Index outer, Index inner) {
812 Index row = rowIndexByOuterInner(outer, inner);
813 Index col = colIndexByOuterInner(outer, inner);
814 assignCoeff(row, col);
815 }
816
817 template <int StoreMode, int LoadMode, typename Packet>
818 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignPacket(Index row, Index col) {
819 m_functor.template assignPacket<StoreMode>(&m_dst.coeffRef(row, col),
820 m_src.template packet<LoadMode, Packet>(row, col));
821 }
822
823 template <int StoreMode, int LoadMode, typename Packet>
824 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignPacket(Index index) {
825 m_functor.template assignPacket<StoreMode>(&m_dst.coeffRef(index), m_src.template packet<LoadMode, Packet>(index));
826 }
827
828 template <int StoreMode, int LoadMode, typename Packet>
829 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignPacketByOuterInner(Index outer, Index inner) {
830 Index row = rowIndexByOuterInner(outer, inner);
831 Index col = colIndexByOuterInner(outer, inner);
832 assignPacket<StoreMode, LoadMode, Packet>(row, col);
833 }
834
835 template <int StoreMode, int LoadMode, typename Packet>
836 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignPacketSegment(Index row, Index col, Index begin, Index count) {
837 m_functor.template assignPacketSegment<StoreMode>(
838 &m_dst.coeffRef(row, col), m_src.template packetSegment<LoadMode, Packet>(row, col, begin, count), begin,
839 count);
840 }
841
842 template <int StoreMode, int LoadMode, typename Packet>
843 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignPacketSegment(Index index, Index begin, Index count) {
844 m_functor.template assignPacketSegment<StoreMode>(
845 &m_dst.coeffRef(index), m_src.template packetSegment<LoadMode, Packet>(index, begin, count), begin, count);
846 }
847
848 template <int StoreMode, int LoadMode, typename Packet>
849 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void assignPacketSegmentByOuterInner(Index outer, Index inner, Index begin,
850 Index count) {
851 Index row = rowIndexByOuterInner(outer, inner);
852 Index col = colIndexByOuterInner(outer, inner);
853 assignPacketSegment<StoreMode, LoadMode, Packet>(row, col, begin, count);
854 }
855
856 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr Index rowIndexByOuterInner(Index outer, Index inner) {
857 using Traits = typename DstEvaluatorType::ExpressionTraits;
858 return int(Traits::RowsAtCompileTime) == 1 ? 0
859 : int(Traits::ColsAtCompileTime) == 1 ? inner
860 : int(DstEvaluatorType::Flags) & RowMajorBit ? outer
861 : inner;
862 }
863
864 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr Index colIndexByOuterInner(Index outer, Index inner) {
865 using Traits = typename DstEvaluatorType::ExpressionTraits;
866 return int(Traits::ColsAtCompileTime) == 1 ? 0
867 : int(Traits::RowsAtCompileTime) == 1 ? inner
868 : int(DstEvaluatorType::Flags) & RowMajorBit ? inner
869 : outer;
870 }
871
872 EIGEN_DEVICE_FUNC const Scalar* dstDataPtr() const { return m_dstExpr.data(); }
873
874 protected:
875 DstEvaluatorType& m_dst;
876 const SrcEvaluatorType& m_src;
877 const Functor& m_functor;
878 // TODO: find a way to avoid the needs of the original expression
879 DstXprType& m_dstExpr;
880};
881
882// Special kernel used when computing small products whose operands have dynamic dimensions. It ensures that the
883// PacketSize used is no larger than 4, thereby increasing the chance that vectorized instructions will be used
884// when computing the product.
885
886template <typename DstEvaluatorTypeT, typename SrcEvaluatorTypeT, typename Functor>
887class restricted_packet_dense_assignment_kernel
888 : public generic_dense_assignment_kernel<DstEvaluatorTypeT, SrcEvaluatorTypeT, Functor, BuiltIn> {
889 protected:
890 using Base = generic_dense_assignment_kernel<DstEvaluatorTypeT, SrcEvaluatorTypeT, Functor, BuiltIn>;
891
892 public:
893 using Scalar = typename Base::Scalar;
894 using DstXprType = typename Base::DstXprType;
895 using AssignmentTraits = copy_using_evaluator_traits<DstEvaluatorTypeT, SrcEvaluatorTypeT, Functor, 4>;
896 using PacketType = typename AssignmentTraits::PacketType;
897
898 EIGEN_DEVICE_FUNC restricted_packet_dense_assignment_kernel(DstEvaluatorTypeT& dst, const SrcEvaluatorTypeT& src,
899 const Functor& func, DstXprType& dstExpr)
900 : Base(dst, src, func, dstExpr) {}
901};
902
903/***************************************************************************
904 * Part 5 : Entry point for dense rectangular assignment
905 ***************************************************************************/
906
907template <typename DstXprType, typename SrcXprType, typename Functor>
908EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE constexpr void resize_if_allowed(DstXprType& dst, const SrcXprType& src,
909 const Functor& /*func*/) {
910 EIGEN_ONLY_USED_FOR_DEBUG(dst);
911 EIGEN_ONLY_USED_FOR_DEBUG(src);
912 eigen_assert(dst.rows() == src.rows() && dst.cols() == src.cols());
913}
914
915template <typename DstXprType, typename SrcXprType, typename T1, typename T2>
916EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE constexpr void resize_if_allowed(DstXprType& dst, const SrcXprType& src,
917 const internal::assign_op<T1, T2>& /*func*/) {
918 Index dstRows = src.rows();
919 Index dstCols = src.cols();
920 if (((dst.rows() != dstRows) || (dst.cols() != dstCols))) {
921#ifdef EIGEN_NO_AUTOMATIC_RESIZING
922 eigen_assert(dst.size() == 0 &&
923 "Size mismatch. Automatic resizing is disabled because EIGEN_NO_AUTOMATIC_RESIZING is defined");
924 if (dst.size() == 0) {
925 dst.resize(dstRows, dstCols);
926 }
927#else
928 dst.resize(dstRows, dstCols);
929 eigen_assert(dst.rows() == dstRows && dst.cols() == dstCols);
930#endif
931 }
932}
933
934template <typename DstXprType, typename SrcXprType, typename Functor>
935EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE constexpr void call_dense_assignment_loop(DstXprType& dst, const SrcXprType& src,
936 const Functor& func) {
937 using DstEvaluatorType = evaluator<DstXprType>;
938 using SrcEvaluatorType = evaluator<SrcXprType>;
939
940 SrcEvaluatorType srcEvaluator(src);
941
942 // NOTE To properly handle A = (A*A.transpose())/s with A rectangular,
943 // we need to resize the destination after the source evaluator has been created.
944 resize_if_allowed(dst, src, func);
945
946 DstEvaluatorType dstEvaluator(dst);
947
948 using Kernel = generic_dense_assignment_kernel<DstEvaluatorType, SrcEvaluatorType, Functor>;
949 Kernel kernel(dstEvaluator, srcEvaluator, func, dst.const_cast_derived());
950
951 dense_assignment_loop<Kernel>::run(kernel);
952}
953
954template <typename DstXprType, typename SrcXprType>
955EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void call_dense_assignment_loop(DstXprType& dst, const SrcXprType& src) {
956 call_dense_assignment_loop(dst, src, internal::assign_op<typename DstXprType::Scalar, typename SrcXprType::Scalar>());
957}
958
959/***************************************************************************
960 * Part 6 : Generic assignment
961 ***************************************************************************/
962
963// Based on the respective shapes of the destination and source,
964// the class AssignmentKind determine the kind of assignment mechanism.
965// AssignmentKind must define a Kind typedef.
966template <typename DstShape, typename SrcShape>
967struct AssignmentKind;
968
969// Assignment kind defined in this file:
970struct Dense2Dense {};
971struct EigenBase2EigenBase {};
972
973template <typename, typename>
974struct AssignmentKind {
975 using Kind = EigenBase2EigenBase;
976};
977template <>
978struct AssignmentKind<DenseShape, DenseShape> {
979 using Kind = Dense2Dense;
980};
981
982// This is the main assignment class
983template <typename DstXprType, typename SrcXprType, typename Functor,
984 typename Kind = typename AssignmentKind<typename evaluator_traits<DstXprType>::Shape,
985 typename evaluator_traits<SrcXprType>::Shape>::Kind,
986 typename EnableIf = void>
987struct Assignment;
988
989// The only purpose of this call_assignment() function is to deal with noalias() / "assume-aliasing" and automatic
990// transposition. Indeed, I (Gael) think that this concept of "assume-aliasing" was a mistake, and it makes thing quite
991// complicated. So this intermediate function removes everything related to "assume-aliasing" such that Assignment does
992// not has to bother about these annoying details.
993
994template <typename Dst, typename Src>
995EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE constexpr void call_assignment(Dst& dst, const Src& src) {
996 call_assignment(dst, src, internal::assign_op<typename Dst::Scalar, typename Src::Scalar>());
997}
998template <typename Dst, typename Src>
999EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void call_assignment(const Dst& dst, const Src& src) {
1000 call_assignment(dst, src, internal::assign_op<typename Dst::Scalar, typename Src::Scalar>());
1001}
1002
1003// Deal with "assume-aliasing"
1004template <typename Dst, typename Src, typename Func, std::enable_if_t<evaluator_assume_aliasing<Src>::value, int> = 0>
1005EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE constexpr void call_assignment(Dst& dst, const Src& src, const Func& func) {
1006 typename plain_matrix_type<Src>::type tmp(src);
1007 call_assignment_no_alias(dst, tmp, func);
1008}
1009
1010template <typename Dst, typename Src, typename Func, std::enable_if_t<!evaluator_assume_aliasing<Src>::value, int> = 0>
1011EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE constexpr void call_assignment(Dst& dst, const Src& src, const Func& func) {
1012 call_assignment_no_alias(dst, src, func);
1013}
1014
1015// by-pass "assume-aliasing"
1016// When there is no aliasing, we require that 'dst' has been properly resized
1017template <typename Dst, template <typename> class StorageBase, typename Src, typename Func>
1018EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE constexpr void call_assignment(NoAlias<Dst, StorageBase>& dst, const Src& src,
1019 const Func& func) {
1020 call_assignment_no_alias(dst.expression(), src, func);
1021}
1022
1023template <typename Dst, typename Src, typename Func>
1024EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE constexpr void call_assignment_no_alias(Dst& dst, const Src& src,
1025 const Func& func) {
1026 enum {
1027 NeedToTranspose = ((int(Dst::RowsAtCompileTime) == 1 && int(Src::ColsAtCompileTime) == 1) ||
1028 (int(Dst::ColsAtCompileTime) == 1 && int(Src::RowsAtCompileTime) == 1)) &&
1029 int(Dst::SizeAtCompileTime) != 1
1030 };
1031
1032 using ActualDstTypeCleaned = std::conditional_t<NeedToTranspose, Transpose<Dst>, Dst>;
1033 using ActualDstType = std::conditional_t<NeedToTranspose, Transpose<Dst>, Dst&>;
1034 ActualDstType actualDst(dst);
1035
1036 // TODO: check whether this is the right place to perform these checks:
1037 EIGEN_STATIC_ASSERT_LVALUE(Dst)
1038 EIGEN_STATIC_ASSERT_SAME_MATRIX_SIZE(ActualDstTypeCleaned, Src)
1039 EIGEN_CHECK_BINARY_COMPATIBILITY(Func, typename ActualDstTypeCleaned::Scalar, typename Src::Scalar);
1040
1041 Assignment<ActualDstTypeCleaned, Src, Func>::run(actualDst, src, func);
1042}
1043
1044template <typename Dst, typename Src, typename Func>
1045EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE void call_restricted_packet_assignment_no_alias(Dst& dst, const Src& src,
1046 const Func& func) {
1047 using DstEvaluatorType = evaluator<Dst>;
1048 using SrcEvaluatorType = evaluator<Src>;
1049 using Kernel = restricted_packet_dense_assignment_kernel<DstEvaluatorType, SrcEvaluatorType, Func>;
1050
1051 EIGEN_STATIC_ASSERT_LVALUE(Dst)
1052 EIGEN_CHECK_BINARY_COMPATIBILITY(Func, typename Dst::Scalar, typename Src::Scalar);
1053
1054 SrcEvaluatorType srcEvaluator(src);
1055 resize_if_allowed(dst, src, func);
1056
1057 DstEvaluatorType dstEvaluator(dst);
1058 Kernel kernel(dstEvaluator, srcEvaluator, func, dst.const_cast_derived());
1059
1060 dense_assignment_loop<Kernel>::run(kernel);
1061}
1062
1063template <typename Dst, typename Src>
1064EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE constexpr void call_assignment_no_alias(Dst& dst, const Src& src) {
1065 call_assignment_no_alias(dst, src, internal::assign_op<typename Dst::Scalar, typename Src::Scalar>());
1066}
1067
1068template <typename Dst, typename Src, typename Func>
1069EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE constexpr void call_assignment_no_alias_no_transpose(Dst& dst, const Src& src,
1070 const Func& func) {
1071 // TODO: check whether this is the right place to perform these checks:
1072 EIGEN_STATIC_ASSERT_LVALUE(Dst)
1073 EIGEN_STATIC_ASSERT_SAME_MATRIX_SIZE(Dst, Src)
1074 EIGEN_CHECK_BINARY_COMPATIBILITY(Func, typename Dst::Scalar, typename Src::Scalar);
1075
1076 Assignment<Dst, Src, Func>::run(dst, src, func);
1077}
1078template <typename Dst, typename Src>
1079EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE constexpr void call_assignment_no_alias_no_transpose(Dst& dst, const Src& src) {
1080 call_assignment_no_alias_no_transpose(dst, src, internal::assign_op<typename Dst::Scalar, typename Src::Scalar>());
1081}
1082
1083// forward declaration
1084template <typename Dst, typename Src>
1085EIGEN_DEVICE_FUNC void check_for_aliasing(const Dst& dst, const Src& src);
1086
1087// Generic Dense to Dense assignment
1088// Note that the last template argument "Weak" is needed to make it possible to perform
1089// both partial specialization+SFINAE without ambiguous specialization
1090template <typename DstXprType, typename SrcXprType, typename Functor, typename Weak>
1091struct Assignment<DstXprType, SrcXprType, Functor, Dense2Dense, Weak> {
1092 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE constexpr void run(DstXprType& dst, const SrcXprType& src,
1093 const Functor& func) {
1094#ifndef EIGEN_NO_DEBUG
1095 if (!internal::is_constant_evaluated()) {
1096 internal::check_for_aliasing(dst, src);
1097 }
1098#endif
1099
1100 call_dense_assignment_loop(dst, src, func);
1101 }
1102};
1103
1104template <typename DstXprType, typename SrcPlainObject, typename Weak>
1105struct Assignment<DstXprType, CwiseNullaryOp<scalar_constant_op<typename DstXprType::Scalar>, SrcPlainObject>,
1106 assign_op<typename DstXprType::Scalar, typename DstXprType::Scalar>, Dense2Dense, Weak> {
1107 using Scalar = typename DstXprType::Scalar;
1108 using NullaryOp = scalar_constant_op<Scalar>;
1109 using SrcXprType = CwiseNullaryOp<NullaryOp, SrcPlainObject>;
1110 using Functor = assign_op<Scalar, Scalar>;
1111 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(DstXprType& dst, const SrcXprType& src,
1112 const Functor& /*func*/) {
1113 eigen_fill_impl<DstXprType>::run(dst, src);
1114 }
1115};
1116
1117template <typename DstXprType, typename SrcPlainObject, typename Weak>
1118struct Assignment<DstXprType, CwiseNullaryOp<scalar_zero_op<typename DstXprType::Scalar>, SrcPlainObject>,
1119 assign_op<typename DstXprType::Scalar, typename DstXprType::Scalar>, Dense2Dense, Weak> {
1120 using Scalar = typename DstXprType::Scalar;
1121 using NullaryOp = scalar_zero_op<Scalar>;
1122 using SrcXprType = CwiseNullaryOp<NullaryOp, SrcPlainObject>;
1123 using Functor = assign_op<Scalar, Scalar>;
1124 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(DstXprType& dst, const SrcXprType& src,
1125 const Functor& /*func*/) {
1126 eigen_zero_impl<DstXprType>::run(dst, src);
1127 }
1128};
1129
1130// Generic assignment through evalTo.
1131// TODO: evaluate whether this generic evalTo-based assignment path is still needed.
1132// Note that the last template argument "Weak" is needed to make it possible to perform
1133// both partial specialization+SFINAE without ambiguous specialization
1134template <typename DstXprType, typename SrcXprType, typename Functor, typename Weak>
1135struct Assignment<DstXprType, SrcXprType, Functor, EigenBase2EigenBase, Weak> {
1136 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(
1137 DstXprType& dst, const SrcXprType& src,
1138 const internal::assign_op<typename DstXprType::Scalar, typename SrcXprType::Scalar>& /*func*/) {
1139 Index dstRows = src.rows();
1140 Index dstCols = src.cols();
1141 if ((dst.rows() != dstRows) || (dst.cols() != dstCols)) dst.resize(dstRows, dstCols);
1142
1143 eigen_assert(dst.rows() == src.rows() && dst.cols() == src.cols());
1144 src.evalTo(dst);
1145 }
1146
1147 // NOTE The following two functions are templated to avoid their instantiation if not needed
1148 // This is needed because some expressions supports evalTo only and/or have 'void' as scalar type.
1149 template <typename SrcScalarType>
1150 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(
1151 DstXprType& dst, const SrcXprType& src,
1152 const internal::add_assign_op<typename DstXprType::Scalar, SrcScalarType>& /*func*/) {
1153 Index dstRows = src.rows();
1154 Index dstCols = src.cols();
1155 if ((dst.rows() != dstRows) || (dst.cols() != dstCols)) dst.resize(dstRows, dstCols);
1156
1157 eigen_assert(dst.rows() == src.rows() && dst.cols() == src.cols());
1158 src.addTo(dst);
1159 }
1160
1161 template <typename SrcScalarType>
1162 EIGEN_DEVICE_FUNC static EIGEN_STRONG_INLINE void run(
1163 DstXprType& dst, const SrcXprType& src,
1164 const internal::sub_assign_op<typename DstXprType::Scalar, SrcScalarType>& /*func*/) {
1165 Index dstRows = src.rows();
1166 Index dstCols = src.cols();
1167 if ((dst.rows() != dstRows) || (dst.cols() != dstCols)) dst.resize(dstRows, dstCols);
1168
1169 eigen_assert(dst.rows() == src.rows() && dst.cols() == src.cols());
1170 src.subTo(dst);
1171 }
1172};
1173
1174} // namespace internal
1175
1176} // end namespace Eigen
1177
1178#endif // EIGEN_ASSIGN_EVALUATOR_H
@ Unaligned
Definition Constants.h:236
constexpr unsigned int ActualPacketAccessBit
Definition Constants.h:109
constexpr unsigned int DirectAccessBit
Definition Constants.h:160
constexpr unsigned int LinearAccessBit
Definition Constants.h:134
constexpr unsigned int RowMajorBit
Definition Constants.h:71