11#ifndef EIGEN_TENSOR_TENSOR_COST_MODEL_H
12#define EIGEN_TENSOR_TENSOR_COST_MODEL_H
15#include "./InternalHeaderCheck.h"
26 template <
typename ArgType>
27 static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE
int MulCost() {
28 return internal::functor_traits<internal::scalar_product_op<ArgType, ArgType> >::Cost;
30 template <
typename ArgType>
31 static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE
int AddCost() {
32 return internal::functor_traits<internal::scalar_sum_op<ArgType> >::Cost;
34 template <
typename ArgType>
35 static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE
int DivCost() {
36 return internal::functor_traits<internal::scalar_quotient_op<ArgType, ArgType> >::Cost;
38 template <
typename ArgType>
39 static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE
int ModCost() {
40 return internal::functor_traits<internal::scalar_mod_op<ArgType> >::Cost;
42 template <
typename SrcType,
typename TargetType>
43 static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE
int CastCost() {
44 return internal::functor_traits<internal::scalar_cast_op<SrcType, TargetType> >::Cost;
47 constexpr EIGEN_DEVICE_FUNC TensorOpCost() =
default;
48 constexpr EIGEN_DEVICE_FUNC TensorOpCost(
double bytes_loaded,
double bytes_stored,
double compute_cycles)
49 : bytes_loaded_(bytes_loaded), bytes_stored_(bytes_stored), compute_cycles_(compute_cycles) {}
51 EIGEN_DEVICE_FUNC TensorOpCost(
double bytes_loaded,
double bytes_stored,
double compute_cycles,
bool vectorized,
53 : bytes_loaded_(bytes_loaded),
54 bytes_stored_(bytes_stored),
55 compute_cycles_(vectorized ? compute_cycles / packet_size : compute_cycles) {
56 eigen_assert(bytes_loaded >= 0 && (numext::isfinite)(bytes_loaded));
57 eigen_assert(bytes_stored >= 0 && (numext::isfinite)(bytes_stored));
58 eigen_assert(compute_cycles >= 0 && (numext::isfinite)(compute_cycles));
61 constexpr EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE
double bytes_loaded()
const {
return bytes_loaded_; }
62 constexpr EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE
double bytes_stored()
const {
return bytes_stored_; }
63 constexpr EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE
double compute_cycles()
const {
return compute_cycles_; }
64 constexpr EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE
double total_bytes()
const {
return bytes_loaded_ + bytes_stored_; }
65 constexpr EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE
double total_cost(
double load_cost,
double store_cost,
66 double compute_cost)
const {
67 return load_cost * bytes_loaded_ + store_cost * bytes_stored_ + compute_cost * compute_cycles_;
72 EIGEN_DEVICE_FUNC
void dropMemoryCost() {
78 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost cwiseMin(
const TensorOpCost& rhs)
const {
79 double bytes_loaded = numext::mini(bytes_loaded_, rhs.bytes_loaded());
80 double bytes_stored = numext::mini(bytes_stored_, rhs.bytes_stored());
81 double compute_cycles = numext::mini(compute_cycles_, rhs.compute_cycles());
82 return TensorOpCost(bytes_loaded, bytes_stored, compute_cycles);
86 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost cwiseMax(
const TensorOpCost& rhs)
const {
87 double bytes_loaded = numext::maxi(bytes_loaded_, rhs.bytes_loaded());
88 double bytes_stored = numext::maxi(bytes_stored_, rhs.bytes_stored());
89 double compute_cycles = numext::maxi(compute_cycles_, rhs.compute_cycles());
90 return TensorOpCost(bytes_loaded, bytes_stored, compute_cycles);
93 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost& operator+=(
const TensorOpCost& rhs) {
94 bytes_loaded_ += rhs.bytes_loaded();
95 bytes_stored_ += rhs.bytes_stored();
96 compute_cycles_ += rhs.compute_cycles();
100 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorOpCost& operator*=(
double rhs) {
101 bytes_loaded_ *= rhs;
102 bytes_stored_ *= rhs;
103 compute_cycles_ *= rhs;
107 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE
friend TensorOpCost operator+(TensorOpCost lhs,
const TensorOpCost& rhs) {
111 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE
friend TensorOpCost operator*(TensorOpCost lhs,
double rhs) {
115 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE
friend TensorOpCost operator*(
double lhs, TensorOpCost rhs) {
120 friend std::ostream& operator<<(std::ostream& os,
const TensorOpCost& tc) {
121 return os <<
"[bytes_loaded = " << tc.bytes_loaded() <<
", bytes_stored = " << tc.bytes_stored()
122 <<
", compute_cycles = " << tc.compute_cycles() <<
"]";
126 double bytes_loaded_ = 0;
127 double bytes_stored_ = 0;
128 double compute_cycles_ = 0;
142template <
typename Device>
146 static constexpr int kDeviceCyclesPerComputeCycle = 1;
150 static constexpr int kStartupCycles = 25000;
152 static constexpr int kPerThreadCycles = 25000;
153 static constexpr int kTaskSize = 40000;
157 static constexpr int kMemBandwidthSaturationThreads = 4;
167 static constexpr double kMemBoundThreshold = 2.0;
172 static constexpr double kDramThresholdBytes = 1024.0 * 1024.0;
177 static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE
int numThreads(
double output_size,
const TensorOpCost& cost_per_coeff,
179 if (max_threads <= 1)
return 1;
181 double mem = memoryTime(cost_per_coeff);
182 double comp = computeTime(cost_per_coeff);
183 double per_coeff = numext::maxi(mem, comp);
184 double total = output_size * per_coeff;
187 if (total < kStartupCycles)
return 1;
191 double threads = total / kPerThreadCycles;
193 threads = numext::mini<double>(threads, GenericNumTraits<int>::highest());
194 int candidate = numext::mini(max_threads, numext::maxi<int>(1,
static_cast<int>(threads)));
205 const int mem_bandwidth_saturation_threads = kMemBandwidthSaturationThreads;
206 if (candidate > mem_bandwidth_saturation_threads) {
207 bool is_memory_bound = (comp > 0) ? (mem / comp > kMemBoundThreshold) : (mem > 0);
208 if (is_memory_bound) {
209 double total_bytes = output_size * cost_per_coeff.total_bytes();
210 if (total_bytes > kDramThresholdBytes) {
211 candidate = numext::mini(candidate, mem_bandwidth_saturation_threads);
221 static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE
double taskSize(
double output_size,
const TensorOpCost& cost_per_coeff) {
222 return totalCost(output_size, cost_per_coeff) / kTaskSize;
226 static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE
double totalCost(
double output_size,
227 const TensorOpCost& cost_per_coeff) {
228 double mem_cost = memoryTime(cost_per_coeff);
229 double comp_cost = computeTime(cost_per_coeff);
230 return output_size * numext::maxi(mem_cost, comp_cost);
238 static constexpr double kByteCost = 1.0 / 16.0;
240 static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE
double memoryTime(
const TensorOpCost& cost) {
241 return cost.total_bytes() * kByteCost;
244 static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE
double computeTime(
const TensorOpCost& cost) {
245 return cost.compute_cycles() * kDeviceCyclesPerComputeCycle;
A cost model used to limit the number of threads used for evaluating tensor expressions.
Definition TensorCostModel.h:143
Namespace containing all symbols from the Eigen library.