Eigen-Contrib  5.0.1
 
Loading...
Searching...
No Matches
TensorPadding.h
1// This file is part of Eigen, a lightweight C++ template library
2// for linear algebra.
3//
4// Copyright (C) 2014 Benoit Steiner <benoit.steiner.goog@gmail.com>
5//
6// This Source Code Form is subject to the terms of the Mozilla
7// Public License v. 2.0. If a copy of the MPL was not distributed
8// with this file, You can obtain one at http://mozilla.org/MPL/2.0/.
9// SPDX-License-Identifier: MPL-2.0
10
11#ifndef EIGEN_TENSOR_TENSOR_PADDING_H
12#define EIGEN_TENSOR_TENSOR_PADDING_H
13
14// IWYU pragma: private
15#include "./InternalHeaderCheck.h"
16
17namespace Eigen {
18
19namespace internal {
20template <typename PaddingDimensions, typename XprType>
21struct traits<TensorPaddingOp<PaddingDimensions, XprType> > : public traits<XprType> {
22 typedef typename XprType::Scalar Scalar;
23 typedef traits<XprType> XprTraits;
24 typedef typename XprTraits::StorageKind StorageKind;
25 typedef typename XprTraits::Index Index;
26 static constexpr int NumDimensions = XprTraits::NumDimensions;
27 static constexpr int Layout = XprTraits::Layout;
28 typedef typename XprTraits::PointerType PointerType;
29};
30
31template <typename PaddingDimensions, typename XprType>
32struct eval<TensorPaddingOp<PaddingDimensions, XprType>, Eigen::Dense> {
33 typedef const TensorPaddingOp<PaddingDimensions, XprType>& type;
34};
35
36} // end namespace internal
37
45template <typename PaddingDimensions, typename XprType>
46class TensorPaddingOp : public TensorBase<TensorPaddingOp<PaddingDimensions, XprType>, ReadOnlyAccessors> {
47 public:
48 typedef typename Eigen::internal::traits<TensorPaddingOp>::Scalar Scalar;
49 typedef typename Eigen::NumTraits<Scalar>::Real RealScalar;
50 typedef typename XprType::CoeffReturnType CoeffReturnType;
51 typedef typename Eigen::internal::ref_selector<TensorPaddingOp>::type Nested;
52 typedef typename Eigen::internal::traits<TensorPaddingOp>::StorageKind StorageKind;
53 typedef typename Eigen::internal::traits<TensorPaddingOp>::Index Index;
54
55 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorPaddingOp(const XprType& expr, const PaddingDimensions& padding_dims,
56 const Scalar padding_value)
57 : m_xpr(expr), m_padding_dims(padding_dims), m_padding_value(padding_value) {}
58
59 EIGEN_DEVICE_FUNC const PaddingDimensions& padding() const { return m_padding_dims; }
60 EIGEN_DEVICE_FUNC Scalar padding_value() const { return m_padding_value; }
61
62 EIGEN_DEVICE_FUNC const internal::remove_all_t<typename XprType::Nested>& expression() const { return m_xpr; }
63
64 protected:
65 typename XprType::Nested m_xpr;
66 const PaddingDimensions m_padding_dims;
67 const Scalar m_padding_value;
68};
69
70// Eval as rvalue
71template <typename PaddingDimensions, typename ArgType, typename Device>
72struct TensorEvaluator<const TensorPaddingOp<PaddingDimensions, ArgType>, Device> {
74 typedef typename XprType::Index Index;
75 static constexpr int NumDims = internal::array_size<PaddingDimensions>::value;
76 typedef DSizes<Index, NumDims> Dimensions;
77 typedef typename XprType::Scalar Scalar;
78 typedef typename XprType::CoeffReturnType CoeffReturnType;
79 typedef typename PacketType<CoeffReturnType, Device>::type PacketReturnType;
80 static constexpr int PacketSize = PacketType<CoeffReturnType, Device>::size;
81 typedef StorageMemory<CoeffReturnType, Device> Storage;
82 typedef typename Storage::Type EvaluatorPointerType;
83
84 static constexpr int Layout = TensorEvaluator<ArgType, Device>::Layout;
85 enum {
86 IsAligned = true,
87 PacketAccess = TensorEvaluator<ArgType, Device>::PacketAccess,
89 PreferBlockAccess = true,
90 CoordAccess = true,
91 RawAccess = false
92 };
93
94 typedef std::remove_const_t<Scalar> ScalarNoConst;
95
96 //===- Tensor block evaluation strategy (see TensorBlock.h) -------------===//
97 typedef internal::TensorBlockDescriptor<NumDims, Index> TensorBlockDesc;
98 typedef internal::TensorBlockScratchAllocator<Device> TensorBlockScratch;
99
100 typedef typename internal::TensorMaterializedBlock<ScalarNoConst, NumDims, Layout, Index> TensorBlock;
101 //===--------------------------------------------------------------------===//
102
103 EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device)
104 : m_impl(op.expression(), device), m_padding(op.padding()), m_paddingValue(op.padding_value()), m_device(device) {
105 // The padding op doesn't change the rank of the tensor. Directly padding a scalar would lead
106 // to a vector, which doesn't make sense. Instead one should reshape the scalar into a vector
107 // of 1 element first and then pad.
108 EIGEN_STATIC_ASSERT((NumDims > 0), YOU_MADE_A_PROGRAMMING_MISTAKE);
109
110 // Compute dimensions
111 m_dimensions = m_impl.dimensions();
112 for (int i = 0; i < NumDims; ++i) {
113 m_dimensions[i] += m_padding[i].first + m_padding[i].second;
114 }
115 const typename TensorEvaluator<ArgType, Device>::Dimensions& input_dims = m_impl.dimensions();
116 EIGEN_IF_CONSTEXPR (static_cast<int>(Layout) == static_cast<int>(ColMajor)) {
117 m_inputStrides[0] = 1;
118 m_outputStrides[0] = 1;
119 for (int i = 1; i < NumDims; ++i) {
120 m_inputStrides[i] = m_inputStrides[i - 1] * input_dims[i - 1];
121 m_outputStrides[i] = m_outputStrides[i - 1] * m_dimensions[i - 1];
122 }
123 m_outputStrides[NumDims] = m_outputStrides[NumDims - 1] * m_dimensions[NumDims - 1];
124 } else {
125 m_inputStrides[NumDims - 1] = 1;
126 m_outputStrides[NumDims] = 1;
127 for (int i = NumDims - 2; i >= 0; --i) {
128 m_inputStrides[i] = m_inputStrides[i + 1] * input_dims[i + 1];
129 m_outputStrides[i + 1] = m_outputStrides[i + 2] * m_dimensions[i + 1];
130 }
131 m_outputStrides[0] = m_outputStrides[1] * m_dimensions[0];
132 }
133 }
134
135 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE const Dimensions& dimensions() const { return m_dimensions; }
136
137 EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(EvaluatorPointerType) {
138 m_impl.evalSubExprsIfNeeded(nullptr);
139 return true;
140 }
141
142#ifdef EIGEN_USE_THREADS
143 template <typename EvalSubExprsCallback>
144 EIGEN_STRONG_INLINE void evalSubExprsIfNeededAsync(EvaluatorPointerType, EvalSubExprsCallback done) {
145 m_impl.evalSubExprsIfNeededAsync(nullptr, [done](bool) { done(true); });
146 }
147#endif // EIGEN_USE_THREADS
148
149 EIGEN_STRONG_INLINE void cleanup() { m_impl.cleanup(); }
150
151 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE CoeffReturnType coeff(Index index) const {
152 eigen_assert(index < dimensions().TotalSize());
153 Index inputIndex = 0;
154 EIGEN_IF_CONSTEXPR (static_cast<int>(Layout) == static_cast<int>(ColMajor)) {
155 EIGEN_UNROLL_LOOP
156 for (int i = NumDims - 1; i > 0; --i) {
157 const Index idx = index / m_outputStrides[i];
158 if (isPaddingAtIndexForDim(idx, i)) {
159 return m_paddingValue;
160 }
161 inputIndex += (idx - m_padding[i].first) * m_inputStrides[i];
162 index -= idx * m_outputStrides[i];
163 }
164 if (isPaddingAtIndexForDim(index, 0)) {
165 return m_paddingValue;
166 }
167 inputIndex += (index - m_padding[0].first);
168 } else {
169 EIGEN_UNROLL_LOOP
170 for (int i = 0; i < NumDims - 1; ++i) {
171 const Index idx = index / m_outputStrides[i + 1];
172 if (isPaddingAtIndexForDim(idx, i)) {
173 return m_paddingValue;
174 }
175 inputIndex += (idx - m_padding[i].first) * m_inputStrides[i];
176 index -= idx * m_outputStrides[i + 1];
177 }
178 if (isPaddingAtIndexForDim(index, NumDims - 1)) {
179 return m_paddingValue;
180 }
181 inputIndex += (index - m_padding[NumDims - 1].first);
182 }
183 return m_impl.coeff(inputIndex);
184 }
185
186 template <int LoadMode>
187 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packet(Index index) const {
188 EIGEN_IF_CONSTEXPR (static_cast<int>(Layout) == static_cast<int>(ColMajor)) {
189 return packetColMajor(index);
190 }
191 return packetRowMajor(index);
192 }
193
194 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost costPerCoeff(bool vectorized) const {
195 TensorOpCost cost = m_impl.costPerCoeff(vectorized);
196 EIGEN_IF_CONSTEXPR (static_cast<int>(Layout) == static_cast<int>(ColMajor)) {
197 EIGEN_UNROLL_LOOP
198 for (int i = 0; i < NumDims; ++i) updateCostPerDimension(cost, i, i == 0);
199 } else {
200 EIGEN_UNROLL_LOOP
201 for (int i = NumDims - 1; i >= 0; --i) updateCostPerDimension(cost, i, i == NumDims - 1);
202 }
203 return cost;
204 }
205
206 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE internal::TensorBlockResourceRequirements getResourceRequirements() const {
207 const size_t target_size = m_device.lastLevelCacheSize();
208 return internal::TensorBlockResourceRequirements::merge(
209 internal::TensorBlockResourceRequirements::skewed<Scalar>(target_size), m_impl.getResourceRequirements());
210 }
211
212 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorBlock block(TensorBlockDesc& desc, TensorBlockScratch& scratch,
213 bool /*root_of_expr_ast*/ = false) const {
214 // If one of the dimensions is zero, return empty block view.
215 if (desc.size() == 0) {
216 return TensorBlock(internal::TensorBlockKind::kView, nullptr, desc.dimensions());
217 }
218
219 static constexpr bool IsColMajor = Layout == static_cast<int>(ColMajor);
220 constexpr int inner_dim_idx = IsColMajor ? 0 : NumDims - 1;
221
222 Index offset = desc.offset();
223
224 // Compute offsets in the output tensor corresponding to the desc.offset().
225 DSizes<Index, NumDims> output_offsets;
226 for (int i = NumDims - 1; i > 0; --i) {
227 const int dim = IsColMajor ? i : NumDims - i - 1;
228 const int stride_dim = IsColMajor ? dim : dim + 1;
229 output_offsets[dim] = offset / m_outputStrides[stride_dim];
230 offset -= output_offsets[dim] * m_outputStrides[stride_dim];
231 }
232 output_offsets[inner_dim_idx] = offset;
233
234 // Offsets in the input corresponding to output offsets.
235 DSizes<Index, NumDims> input_offsets = output_offsets;
236 for (int i = 0; i < NumDims; ++i) {
237 const int dim = IsColMajor ? i : NumDims - i - 1;
238 input_offsets[dim] = input_offsets[dim] - m_padding[dim].first;
239 }
240
241 // Compute offset in the input buffer (at this point it might be illegal and
242 // point outside of the input buffer, because we don't check for negative
243 // offsets, it will be autocorrected in the block iteration loop below).
244 Index input_offset = 0;
245 for (int i = 0; i < NumDims; ++i) {
246 const int dim = IsColMajor ? i : NumDims - i - 1;
247 input_offset += input_offsets[dim] * m_inputStrides[dim];
248 }
249
250 // Destination buffer and scratch buffer both indexed from 0 and have the
251 // same dimensions as the requested block (for destination buffer this
252 // property is guaranteed by `desc.destination()`).
253 Index output_offset = 0;
254 const DSizes<Index, NumDims> output_strides = internal::strides<Layout>(desc.dimensions());
255
256 // NOTE(ezhulenev): We initialize block iteration state for `NumDims - 1`
257 // dimensions, skipping innermost dimension. In theory it should be possible
258 // to squeeze matching innermost dimensions, however in practice that did
259 // not show any improvements in benchmarks. Also in practice first outer
260 // dimension usually has padding, and will prevent squeezing.
261
262 // Initialize output block iterator state. Dimension in this array are
263 // always in inner_most -> outer_most order (col major layout).
264 array<BlockIteratorState, NumDims - 1> it;
265 for (int i = 0; i < NumDims - 1; ++i) {
266 const int dim = IsColMajor ? i + 1 : NumDims - i - 2;
267 it[i].count = 0;
268 it[i].size = desc.dimension(dim);
269
270 it[i].input_stride = m_inputStrides[dim];
271 it[i].input_span = it[i].input_stride * (it[i].size - 1);
272
273 it[i].output_stride = output_strides[dim];
274 it[i].output_span = it[i].output_stride * (it[i].size - 1);
275 }
276
277 const Index input_inner_dim_size = static_cast<Index>(m_impl.dimensions()[inner_dim_idx]);
278
279 // Total output size.
280 const Index output_size = desc.size();
281
282 // We will fill inner dimension of this size in the output. It might be
283 // larger than the inner dimension in the input, so we might have to pad
284 // before/after we copy values from the input inner dimension.
285 const Index output_inner_dim_size = desc.dimension(inner_dim_idx);
286
287 // How many values to fill with padding BEFORE reading from the input inner
288 // dimension.
289 const Index output_inner_pad_before_size =
290 input_offsets[inner_dim_idx] < 0
291 ? numext::mini(numext::abs(input_offsets[inner_dim_idx]), output_inner_dim_size)
292 : 0;
293
294 // How many values we can actually copy from the input inner dimension.
295 const Index output_inner_copy_size = numext::mini(
296 // Want to copy from input.
297 (output_inner_dim_size - output_inner_pad_before_size),
298 // Can copy from input.
299 numext::maxi(input_inner_dim_size - (input_offsets[inner_dim_idx] + output_inner_pad_before_size), Index(0)));
300
301 eigen_assert(output_inner_copy_size >= 0);
302
303 // How many values to fill with padding AFTER reading from the input inner
304 // dimension.
305 const Index output_inner_pad_after_size =
306 output_inner_dim_size - output_inner_copy_size - output_inner_pad_before_size;
307
308 // Sanity check, sum of all sizes must be equal to the output size.
309 eigen_assert(output_inner_dim_size ==
310 (output_inner_pad_before_size + output_inner_copy_size + output_inner_pad_after_size));
311
312 // Keep track of current coordinates and padding in the output.
313 DSizes<Index, NumDims> output_coord = output_offsets;
314 DSizes<Index, NumDims> output_padded;
315 for (int i = 0; i < NumDims; ++i) {
316 const int dim = IsColMajor ? i : NumDims - i - 1;
317 output_padded[dim] = isPaddingAtIndexForDim(output_coord[dim], dim);
318 }
319
320 typedef internal::StridedLinearBufferCopy<ScalarNoConst, Index> LinCopy;
321
322 // Prepare storage for the materialized padding result.
323 const typename TensorBlock::Storage block_storage = TensorBlock::prepareStorage(desc, scratch);
324
325 // TODO(ezhulenev): Squeeze multiple non-padded inner dimensions into a
326 // single logical inner dimension.
327
328 // When possible we squeeze writes for the innermost (only if non-padded)
329 // dimension with the first padded dimension. This allows to reduce the
330 // number of calls to LinCopy and better utilize vector instructions.
331 const bool squeeze_writes = NumDims > 1 &&
332 // inner dimension is not padded
333 (input_inner_dim_size == m_dimensions[inner_dim_idx]) &&
334 // and equal to the block inner dimension
335 (input_inner_dim_size == output_inner_dim_size);
336
337 constexpr int squeeze_dim = NumDims > 1 ? (IsColMajor ? inner_dim_idx + 1 : inner_dim_idx - 1) : 0;
338
339 // Maximum coordinate on a squeeze dimension that we can write to.
340 const Index squeeze_max_coord =
341 squeeze_writes ? numext::mini(
342 // max non-padded element in the input
343 static_cast<Index>(m_dimensions[squeeze_dim] - m_padding[squeeze_dim].second),
344 // max element in the output buffer
345 static_cast<Index>(output_offsets[squeeze_dim] + desc.dimension(squeeze_dim)))
346 : static_cast<Index>(0);
347
348 // Iterate copying data from `m_impl.data()` to the output buffer.
349 for (Index size = 0; size < output_size;) {
350 // Detect if we are in the padded region (exclude innermost dimension).
351 bool is_padded = false;
352 for (int j = 1; j < NumDims; ++j) {
353 const int dim = IsColMajor ? j : NumDims - j - 1;
354 is_padded = output_padded[dim];
355 if (is_padded) break;
356 }
357
358 if (is_padded) {
359 // Fill single innermost dimension with padding value.
360 size += output_inner_dim_size;
361
362 LinCopy::template Run<LinCopy::Kind::FillLinear>(typename LinCopy::Dst(output_offset, 1, block_storage.data()),
363 typename LinCopy::Src(0, 0, &m_paddingValue),
364 output_inner_dim_size);
365
366 } else if (squeeze_writes) {
367 // Squeeze multiple reads from innermost dimensions.
368 const Index squeeze_num = squeeze_max_coord - output_coord[squeeze_dim];
369 size += output_inner_dim_size * squeeze_num;
370
371 // Copy `squeeze_num` inner dimensions from input to output.
372 LinCopy::template Run<LinCopy::Kind::Linear>(typename LinCopy::Dst(output_offset, 1, block_storage.data()),
373 typename LinCopy::Src(input_offset, 1, m_impl.data()),
374 output_inner_dim_size * squeeze_num);
375
376 // Update iteration state for only `squeeze_num - 1` processed inner
377 // dimensions, because we have another iteration state update at the end
378 // of the loop that will update iteration state for the last inner
379 // processed dimension.
380 it[0].count += (squeeze_num - 1);
381 input_offset += it[0].input_stride * (squeeze_num - 1);
382 output_offset += it[0].output_stride * (squeeze_num - 1);
383 output_coord[squeeze_dim] += (squeeze_num - 1);
384
385 } else {
386 // Single read from innermost dimension.
387 size += output_inner_dim_size;
388
389 { // Fill with padding before copying from input inner dimension.
390 const Index out = output_offset;
391
392 LinCopy::template Run<LinCopy::Kind::FillLinear>(typename LinCopy::Dst(out, 1, block_storage.data()),
393 typename LinCopy::Src(0, 0, &m_paddingValue),
394 output_inner_pad_before_size);
395 }
396
397 { // Copy data from input inner dimension.
398 const Index out = output_offset + output_inner_pad_before_size;
399 const Index in = input_offset + output_inner_pad_before_size;
400
401 eigen_assert(output_inner_copy_size == 0 || m_impl.data() != nullptr);
402
403 LinCopy::template Run<LinCopy::Kind::Linear>(typename LinCopy::Dst(out, 1, block_storage.data()),
404 typename LinCopy::Src(in, 1, m_impl.data()),
405 output_inner_copy_size);
406 }
407
408 { // Fill with padding after copying from input inner dimension.
409 const Index out = output_offset + output_inner_pad_before_size + output_inner_copy_size;
410
411 LinCopy::template Run<LinCopy::Kind::FillLinear>(typename LinCopy::Dst(out, 1, block_storage.data()),
412 typename LinCopy::Src(0, 0, &m_paddingValue),
413 output_inner_pad_after_size);
414 }
415 }
416
417 for (int j = 0; j < NumDims - 1; ++j) {
418 const int dim = IsColMajor ? j + 1 : NumDims - j - 2;
419
420 if (++it[j].count < it[j].size) {
421 input_offset += it[j].input_stride;
422 output_offset += it[j].output_stride;
423 output_coord[dim] += 1;
424 output_padded[dim] = isPaddingAtIndexForDim(output_coord[dim], dim);
425 break;
426 }
427 it[j].count = 0;
428 input_offset -= it[j].input_span;
429 output_offset -= it[j].output_span;
430 output_coord[dim] -= it[j].size - 1;
431 output_padded[dim] = isPaddingAtIndexForDim(output_coord[dim], dim);
432 }
433 }
434
435 return block_storage.AsTensorMaterializedBlock();
436 }
437
438 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE EvaluatorPointerType data() const { return nullptr; }
439
440 private:
441 struct BlockIteratorState {
442 BlockIteratorState() = default;
443
444 Index count = 0;
445 Index size = 0;
446 Index input_stride = 0;
447 Index input_span = 0;
448 Index output_stride = 0;
449 Index output_span = 0;
450 };
451
452 EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE bool isPaddingAtIndexForDim(Index index, int dim_index) const {
453 return (!internal::index_pair_first_statically_eq<PaddingDimensions>(dim_index, 0) &&
454 index < m_padding[dim_index].first) ||
455 (!internal::index_pair_second_statically_eq<PaddingDimensions>(dim_index, 0) &&
456 index >= m_dimensions[dim_index] - m_padding[dim_index].second);
457 }
458
459 static EIGEN_DEVICE_FUNC constexpr bool isLeftPaddingCompileTimeZero(int dim_index) {
460 return internal::index_pair_first_statically_eq<PaddingDimensions>(dim_index, 0);
461 }
462
463 static EIGEN_DEVICE_FUNC constexpr bool isRightPaddingCompileTimeZero(int dim_index) {
464 return internal::index_pair_second_statically_eq<PaddingDimensions>(dim_index, 0);
465 }
466
467 void updateCostPerDimension(TensorOpCost& cost, int i, bool first) const {
468 const double in = static_cast<double>(m_impl.dimensions()[i]);
469 const double out = in + m_padding[i].first + m_padding[i].second;
470 if (out == 0) return;
471 const double reduction = in / out;
472 cost *= reduction;
473 if (first) {
474 cost += TensorOpCost(0, 0, 2 * TensorOpCost::AddCost<Index>() + reduction * (1 * TensorOpCost::AddCost<Index>()));
475 } else {
476 cost += TensorOpCost(0, 0,
477 2 * TensorOpCost::AddCost<Index>() + 2 * TensorOpCost::MulCost<Index>() +
478 reduction * (2 * TensorOpCost::MulCost<Index>() + 1 * TensorOpCost::DivCost<Index>()));
479 }
480 }
481
482 protected:
483 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packetColMajor(Index index) const {
484 eigen_assert(index + PacketSize - 1 < dimensions().TotalSize());
485
486 const Index initialIndex = index;
487 Index inputIndex = 0;
488 EIGEN_UNROLL_LOOP
489 for (int i = NumDims - 1; i > 0; --i) {
490 const Index firstIdx = index;
491 const Index lastIdx = index + PacketSize - 1;
492 const Index lastPaddedLeft = m_padding[i].first * m_outputStrides[i];
493 const Index firstPaddedRight = (m_dimensions[i] - m_padding[i].second) * m_outputStrides[i];
494 const Index lastPaddedRight = m_outputStrides[i + 1];
495
496 if (!isLeftPaddingCompileTimeZero(i) && lastIdx < lastPaddedLeft) {
497 // all the coefficients are in the padding zone.
498 return internal::pset1<PacketReturnType>(m_paddingValue);
499 } else if (!isRightPaddingCompileTimeZero(i) && firstIdx >= firstPaddedRight && lastIdx < lastPaddedRight) {
500 // all the coefficients are in the padding zone.
501 return internal::pset1<PacketReturnType>(m_paddingValue);
502 } else if ((isLeftPaddingCompileTimeZero(i) && isRightPaddingCompileTimeZero(i)) ||
503 (firstIdx >= lastPaddedLeft && lastIdx < firstPaddedRight)) {
504 // all the coefficients are between the 2 padding zones.
505 const Index idx = index / m_outputStrides[i];
506 inputIndex += (idx - m_padding[i].first) * m_inputStrides[i];
507 index -= idx * m_outputStrides[i];
508 } else {
509 // Every other case
510 return packetWithPossibleZero(initialIndex);
511 }
512 }
513
514 const Index lastIdx = index + PacketSize - 1;
515 const Index firstIdx = index;
516 const Index lastPaddedLeft = m_padding[0].first;
517 const Index firstPaddedRight = m_dimensions[0] - m_padding[0].second;
518 const Index lastPaddedRight = m_outputStrides[1];
519
520 EIGEN_IF_CONSTEXPR (!isLeftPaddingCompileTimeZero(0)) {
521 if (lastIdx < lastPaddedLeft) {
522 // all the coefficients are in the padding zone.
523 return internal::pset1<PacketReturnType>(m_paddingValue);
524 }
525 }
526 EIGEN_IF_CONSTEXPR (!isRightPaddingCompileTimeZero(0)) {
527 if (firstIdx >= firstPaddedRight && lastIdx < lastPaddedRight) {
528 // all the coefficients are in the padding zone.
529 return internal::pset1<PacketReturnType>(m_paddingValue);
530 }
531 }
532 EIGEN_IF_CONSTEXPR (isLeftPaddingCompileTimeZero(0) && isRightPaddingCompileTimeZero(0)) {
533 // all the coefficients are between the 2 padding zones.
534 inputIndex += (index - m_padding[0].first);
535 return m_impl.template packet<Unaligned>(inputIndex);
536 } else if (firstIdx >= lastPaddedLeft && lastIdx < firstPaddedRight) {
537 // all the coefficients are between the 2 padding zones.
538 inputIndex += (index - m_padding[0].first);
539 return m_impl.template packet<Unaligned>(inputIndex);
540 }
541 // Every other case
542 return packetWithPossibleZero(initialIndex);
543 }
544
545 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packetRowMajor(Index index) const {
546 eigen_assert(index + PacketSize - 1 < dimensions().TotalSize());
547
548 const Index initialIndex = index;
549 Index inputIndex = 0;
550 EIGEN_UNROLL_LOOP
551 for (int i = 0; i < NumDims - 1; ++i) {
552 const Index firstIdx = index;
553 const Index lastIdx = index + PacketSize - 1;
554 const Index lastPaddedLeft = m_padding[i].first * m_outputStrides[i + 1];
555 const Index firstPaddedRight = (m_dimensions[i] - m_padding[i].second) * m_outputStrides[i + 1];
556 const Index lastPaddedRight = m_outputStrides[i];
557
558 if (!isLeftPaddingCompileTimeZero(i) && lastIdx < lastPaddedLeft) {
559 // all the coefficients are in the padding zone.
560 return internal::pset1<PacketReturnType>(m_paddingValue);
561 } else if (!isRightPaddingCompileTimeZero(i) && firstIdx >= firstPaddedRight && lastIdx < lastPaddedRight) {
562 // all the coefficients are in the padding zone.
563 return internal::pset1<PacketReturnType>(m_paddingValue);
564 } else if ((isLeftPaddingCompileTimeZero(i) && isRightPaddingCompileTimeZero(i)) ||
565 (firstIdx >= lastPaddedLeft && lastIdx < firstPaddedRight)) {
566 // all the coefficients are between the 2 padding zones.
567 const Index idx = index / m_outputStrides[i + 1];
568 inputIndex += (idx - m_padding[i].first) * m_inputStrides[i];
569 index -= idx * m_outputStrides[i + 1];
570 } else {
571 // Every other case
572 return packetWithPossibleZero(initialIndex);
573 }
574 }
575
576 const Index lastIdx = index + PacketSize - 1;
577 const Index firstIdx = index;
578 const Index lastPaddedLeft = m_padding[NumDims - 1].first;
579 const Index firstPaddedRight = m_dimensions[NumDims - 1] - m_padding[NumDims - 1].second;
580 const Index lastPaddedRight = m_outputStrides[NumDims - 1];
581
582 EIGEN_IF_CONSTEXPR (!isLeftPaddingCompileTimeZero(NumDims - 1)) {
583 if (lastIdx < lastPaddedLeft) {
584 // all the coefficients are in the padding zone.
585 return internal::pset1<PacketReturnType>(m_paddingValue);
586 }
587 }
588 EIGEN_IF_CONSTEXPR (!isRightPaddingCompileTimeZero(NumDims - 1)) {
589 if (firstIdx >= firstPaddedRight && lastIdx < lastPaddedRight) {
590 // all the coefficients are in the padding zone.
591 return internal::pset1<PacketReturnType>(m_paddingValue);
592 }
593 }
594 EIGEN_IF_CONSTEXPR (isLeftPaddingCompileTimeZero(NumDims - 1) && isRightPaddingCompileTimeZero(NumDims - 1)) {
595 // all the coefficients are between the 2 padding zones.
596 inputIndex += (index - m_padding[NumDims - 1].first);
597 return m_impl.template packet<Unaligned>(inputIndex);
598 } else if (firstIdx >= lastPaddedLeft && lastIdx < firstPaddedRight) {
599 // all the coefficients are between the 2 padding zones.
600 inputIndex += (index - m_padding[NumDims - 1].first);
601 return m_impl.template packet<Unaligned>(inputIndex);
602 }
603 // Every other case
604 return packetWithPossibleZero(initialIndex);
605 }
606
607 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE PacketReturnType packetWithPossibleZero(Index index) const {
608 EIGEN_ALIGN_TO_BOUNDARY(internal::unpacket_traits<PacketReturnType>::alignment)
609 std::remove_const_t<CoeffReturnType> values[PacketSize];
610 EIGEN_UNROLL_LOOP
611 for (int i = 0; i < PacketSize; ++i) {
612 values[i] = coeff(index + i);
613 }
614 PacketReturnType rslt = internal::pload<PacketReturnType>(values);
615 return rslt;
616 }
617
618 Dimensions m_dimensions;
619 array<Index, NumDims + 1> m_outputStrides;
620 array<Index, NumDims> m_inputStrides;
621 TensorEvaluator<ArgType, Device> m_impl;
622 PaddingDimensions m_padding;
623
624 Scalar m_paddingValue;
625
626 const Device EIGEN_DEVICE_REF m_device;
627};
628
629} // end namespace Eigen
630
631#endif // EIGEN_TENSOR_TENSOR_PADDING_H
The tensor base class.
Definition TensorForwardDeclarations.h:69
Tensor padding class. At the moment only padding with a constant value is supported.
Definition TensorPadding.h:46
Namespace containing all symbols from the Eigen library.
The tensor evaluator class.
Definition TensorEvaluator.h:47