11#if defined(EIGEN_USE_GPU) && !defined(EIGEN_TENSOR_TENSOR_DEVICE_GPU_H)
12#define EIGEN_TENSOR_TENSOR_DEVICE_GPU_H
15#include "./InternalHeaderCheck.h"
17#include "../../../../Eigen/src/Core/util/GpuHipCudaDefines.inc"
18#include "../../../../Eigen/src/Core/util/GpuRuntime.h"
22static const int kGpuScratchSize = 1024;
28struct GpuDeviceAttributes {
29 int multiProcessorCount;
30 int maxThreadsPerBlock;
31 int maxThreadsPerMultiProcessor;
32 int sharedMemPerBlock;
33 int sharedMemPerBlockOptin;
34 int computeCapabilityMajor;
35 int computeCapabilityMinor;
37 int memoryPoolsSupported;
44template <
typename GpuDeviceAttr>
45inline int GetGpuDeviceAttribute(GpuDeviceAttr attribute,
int device) {
46#if !defined(EIGEN_USE_HIP)
47 if (attribute == gpuDevAttrMemoryPoolsSupported) {
48 int driver_version = 0;
49 EIGEN_GPU_RUNTIME_CHECK(cudaDriverGetVersion(&driver_version));
51 if (driver_version < 11020)
return 0;
55 const gpuError_t error = gpuDeviceGetAttribute(&value, attribute, device);
56#if defined(EIGEN_USE_HIP)
58 if (attribute == gpuDevAttrMaxSharedMemoryPerBlockOptin && error == hipErrorInvalidValue) {
59 (void)hipGetLastError();
60 EIGEN_GPU_RUNTIME_CHECK(gpuDeviceGetAttribute(&value, gpuDevAttrMaxSharedMemoryPerBlock, device));
64 EIGEN_GPU_RUNTIME_CHECK(error);
68inline const std::vector<GpuDeviceAttributes>& GetGpuDeviceAttributes() {
70 static const std::vector<GpuDeviceAttributes>* kAttributes = [] {
72 EIGEN_GPU_RUNTIME_CHECK(gpuGetDeviceCount(&num_devices));
73 auto* attributes =
new std::vector<GpuDeviceAttributes>(num_devices);
74 for (
int device = 0; device < num_devices; ++device) {
75 (*attributes)[device] = {
76 GetGpuDeviceAttribute(gpuDevAttrMultiProcessorCount, device),
77 GetGpuDeviceAttribute(gpuDevAttrMaxThreadsPerBlock, device),
78 GetGpuDeviceAttribute(gpuDevAttrMaxThreadsPerMultiProcessor, device),
79 GetGpuDeviceAttribute(gpuDevAttrMaxSharedMemoryPerBlock, device),
80 GetGpuDeviceAttribute(gpuDevAttrMaxSharedMemoryPerBlockOptin, device),
81 GetGpuDeviceAttribute(gpuDevAttrComputeCapabilityMajor, device),
82 GetGpuDeviceAttribute(gpuDevAttrComputeCapabilityMinor, device),
83 GetGpuDeviceAttribute(gpuDevAttrWarpSize, device),
84 GetGpuDeviceAttribute(gpuDevAttrMemoryPoolsSupported, device),
92inline const GpuDeviceAttributes& GetGpuDeviceAttributes(
int device) {
93 const std::vector<GpuDeviceAttributes>& attributes = GetGpuDeviceAttributes();
94 eigen_assert(device >= 0 && device <
static_cast<int>(attributes.size()) &&
"no such GPU device");
95 return attributes[device];
100inline const GpuDeviceAttributes& GetCurrentGpuDeviceAttributes() {
102 EIGEN_GPU_RUNTIME_CHECK(gpuGetDevice(&device));
103 return GetGpuDeviceAttributes(device);
108class StreamInterface {
110 virtual ~StreamInterface() =
default;
112 virtual const gpuStream_t& stream()
const = 0;
113 virtual const gpuDeviceProp_t& deviceProperties()
const = 0;
119 virtual const GpuDeviceAttributes& deviceAttributes()
const {
return GetCurrentGpuDeviceAttributes(); }
122 virtual void* allocate(
size_t num_bytes)
const = 0;
123 virtual void deallocate(
void* buffer)
const = 0;
126 virtual void* scratchpad()
const = 0;
132 virtual unsigned int* semaphore()
const = 0;
135class GpuDeviceProperties {
137 static const GpuDeviceProperties& instance() {
138 static const GpuDeviceProperties* kInstance =
new GpuDeviceProperties();
143 EIGEN_STRONG_INLINE
const gpuDeviceProp_t& get(
int device)
const {
144 eigen_assert(device >= 0 && device <
static_cast<int>(device_properties_.size()) &&
"no such GPU device");
145 return device_properties_[device];
149 GpuDeviceProperties() =
default;
151 static std::vector<gpuDeviceProp_t> GetDeviceProperties() {
153 EIGEN_GPU_RUNTIME_CHECK(gpuGetDeviceCount(&num_devices));
154 std::vector<gpuDeviceProp_t> device_properties(num_devices);
155 for (
int i = 0; i < num_devices; ++i) {
156 EIGEN_GPU_RUNTIME_CHECK(gpuGetDeviceProperties(&device_properties[i], i));
159 return device_properties;
162 std::vector<gpuDeviceProp_t> device_properties_ = GetDeviceProperties();
165EIGEN_ALWAYS_INLINE
const GpuDeviceProperties& GetGpuDeviceProperties() {
return GpuDeviceProperties::instance(); }
167EIGEN_ALWAYS_INLINE
const gpuDeviceProp_t& GetGpuDeviceProperties(
int device) {
168 return GetGpuDeviceProperties().get(device);
171static const gpuStream_t default_stream = gpuStreamDefault;
173class GpuStreamDevice :
public StreamInterface {
176 GpuStreamDevice() : stream_(&default_stream), scratch_(nullptr), semaphore_(nullptr) {
177 EIGEN_GPU_RUNTIME_CHECK(gpuGetDevice(&device_));
180 GpuStreamDevice(
int device) : stream_(&default_stream), device_(device), scratch_(nullptr), semaphore_(nullptr) {}
185 GpuStreamDevice(
const gpuStream_t* stream,
int device = -1)
186 : stream_(stream), device_(device), scratch_(nullptr), semaphore_(nullptr) {
188 EIGEN_GPU_RUNTIME_CHECK(gpuGetDevice(&device_));
191 EIGEN_GPU_RUNTIME_CHECK(gpuGetDeviceCount(&num_devices));
192 EIGEN_UNUSED_VARIABLE(num_devices);
193 gpu_assert(device < num_devices);
198 ~GpuStreamDevice()
override {
200 deallocate(scratch_);
204 const gpuStream_t& stream()
const override {
return *stream_; }
205 const gpuDeviceProp_t& deviceProperties()
const override {
return GetGpuDeviceProperties(device_); }
206 const GpuDeviceAttributes& deviceAttributes()
const override {
return GetGpuDeviceAttributes(device_); }
207 void* allocate(
size_t num_bytes)
const override {
208 EIGEN_GPU_RUNTIME_CHECK(gpuSetDevice(device_));
209 void* result =
nullptr;
210 EIGEN_GPU_RUNTIME_CHECK(gpuMalloc(&result, num_bytes));
211 gpu_assert(result !=
nullptr);
214 void deallocate(
void* buffer)
const override {
215 EIGEN_GPU_RUNTIME_CHECK(gpuSetDevice(device_));
216 gpu_assert(buffer !=
nullptr);
217 EIGEN_GPU_RUNTIME_CHECK(gpuFree(buffer));
220 void* scratchpad()
const override {
221 if (scratch_ ==
nullptr) {
222 scratch_ = allocate(kGpuScratchSize +
sizeof(
unsigned int));
227 unsigned int* semaphore()
const override {
228 if (semaphore_ ==
nullptr) {
229 char* scratch =
static_cast<char*
>(scratchpad()) + kGpuScratchSize;
230 semaphore_ =
reinterpret_cast<unsigned int*
>(scratch);
231 EIGEN_GPU_RUNTIME_CHECK(gpuMemsetAsync(semaphore_, 0,
sizeof(
unsigned int), *stream_));
237 const gpuStream_t* stream_;
239 mutable void* scratch_;
240 mutable unsigned int* semaphore_;
246 explicit GpuDevice(
const StreamInterface* stream) : stream_(stream), max_blocks_(INT_MAX) { eigen_assert(stream); }
248 EIGEN_DEPRECATED
explicit GpuDevice(
const StreamInterface* stream,
int num_blocks)
249 : stream_(stream), max_blocks_(num_blocks) {
250 eigen_assert(stream);
253 EIGEN_STRONG_INLINE
const gpuStream_t& stream()
const {
return stream_->stream(); }
255 EIGEN_STRONG_INLINE
void* allocate(
size_t num_bytes)
const {
return stream_->allocate(num_bytes); }
257 EIGEN_STRONG_INLINE
void deallocate(
void* buffer)
const { stream_->deallocate(buffer); }
259 EIGEN_STRONG_INLINE
void* allocate_temp(
size_t num_bytes)
const {
return stream_->allocate(num_bytes); }
261 EIGEN_STRONG_INLINE
void deallocate_temp(
void* buffer)
const { stream_->deallocate(buffer); }
263 template <
typename Type>
264 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Type get(Type data)
const {
268 EIGEN_STRONG_INLINE
void* scratchpad()
const {
return stream_->scratchpad(); }
270 EIGEN_STRONG_INLINE
unsigned int* semaphore()
const {
return stream_->semaphore(); }
272 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE
void memcpy(
void* dst,
const void* src,
size_t n)
const {
273#ifndef EIGEN_GPU_COMPILE_PHASE
274 EIGEN_GPU_RUNTIME_CHECK(gpuMemcpyAsync(dst, src, n, gpuMemcpyDeviceToDevice, stream_->stream()));
276 EIGEN_UNUSED_VARIABLE(dst);
277 EIGEN_UNUSED_VARIABLE(src);
278 EIGEN_UNUSED_VARIABLE(n);
279 eigen_assert(
false &&
"The default device should be used instead to generate kernel code");
283 EIGEN_STRONG_INLINE
void memcpyHostToDevice(
void* dst,
const void* src,
size_t n)
const {
284 EIGEN_GPU_RUNTIME_CHECK(gpuMemcpyAsync(dst, src, n, gpuMemcpyHostToDevice, stream_->stream()));
287 EIGEN_STRONG_INLINE
void memcpyDeviceToHost(
void* dst,
const void* src,
size_t n)
const {
288 EIGEN_GPU_RUNTIME_CHECK(gpuMemcpyAsync(dst, src, n, gpuMemcpyDeviceToHost, stream_->stream()));
291 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE
void memset(
void* buffer,
int c,
size_t n)
const {
292#ifndef EIGEN_GPU_COMPILE_PHASE
293 EIGEN_GPU_RUNTIME_CHECK(gpuMemsetAsync(buffer, c, n, stream_->stream()));
295 EIGEN_UNUSED_VARIABLE(buffer);
296 EIGEN_UNUSED_VARIABLE(c);
297 EIGEN_UNUSED_VARIABLE(n);
298 eigen_assert(
false &&
"The default device should be used instead to generate kernel code");
302 template <
typename T>
303 EIGEN_STRONG_INLINE
void fill(T* begin, T* end,
const T& value)
const {
304#ifndef EIGEN_GPU_COMPILE_PHASE
305 const size_t count = end - begin;
307 const int value_size =
sizeof(value);
308 char* buffer = (
char*)begin;
309 char* value_bytes = (
char*)(&value);
311 bool use_single_memset =
true;
312 for (
int i = 1; i < value_size; ++i) {
313 if (value_bytes[i] != value_bytes[0]) {
314 use_single_memset =
false;
318 if (use_single_memset) {
319 EIGEN_GPU_RUNTIME_CHECK(gpuMemsetAsync(buffer, value_bytes[0], count *
sizeof(T), stream_->stream()));
321 for (
int b = 0; b < value_size; ++b) {
322 EIGEN_GPU_RUNTIME_CHECK(gpuMemset2DAsync(buffer + b, value_size, value_bytes[b], 1, count, stream_->stream()));
326 EIGEN_UNUSED_VARIABLE(begin);
327 EIGEN_UNUSED_VARIABLE(end);
328 EIGEN_UNUSED_VARIABLE(value);
329 eigen_assert(
false &&
"The default device should be used instead to generate kernel code");
334 EIGEN_STRONG_INLINE
size_t numThreads()
const {
return static_cast<size_t>(warpSize()); }
336 EIGEN_STRONG_INLINE
size_t firstLevelCacheSize()
const {
341 EIGEN_STRONG_INLINE
size_t lastLevelCacheSize()
const {
344 return firstLevelCacheSize();
347 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE
void synchronize()
const {
348#ifndef EIGEN_GPU_COMPILE_PHASE
349 EIGEN_GPU_RUNTIME_CHECK(gpuStreamSynchronize(stream_->stream()));
351 gpu_assert(
false &&
"The default device should be used instead to generate kernel code");
355 EIGEN_STRONG_INLINE
int getNumGpuMultiProcessors()
const {
return stream_->deviceProperties().multiProcessorCount; }
356 EIGEN_STRONG_INLINE
int maxGpuThreadsPerBlock()
const {
return stream_->deviceProperties().maxThreadsPerBlock; }
357 EIGEN_STRONG_INLINE
int maxGpuThreadsPerMultiProcessor()
const {
358 return stream_->deviceProperties().maxThreadsPerMultiProcessor;
360 EIGEN_STRONG_INLINE
int sharedMemPerBlock()
const {
361 return static_cast<int>(stream_->deviceProperties().sharedMemPerBlock);
363 EIGEN_STRONG_INLINE
int majorDeviceVersion()
const {
return stream_->deviceProperties().major; }
364 EIGEN_STRONG_INLINE
int minorDeviceVersion()
const {
return stream_->deviceProperties().minor; }
369 EIGEN_STRONG_INLINE
int warpSize()
const {
return stream_->deviceAttributes().warpSize; }
372 EIGEN_STRONG_INLINE
int sharedMemPerBlockOptin()
const {
return stream_->deviceAttributes().sharedMemPerBlockOptin; }
374 EIGEN_STRONG_INLINE
bool memoryPoolsSupported()
const {
375 return stream_->deviceAttributes().memoryPoolsSupported != 0;
378 EIGEN_DEPRECATED EIGEN_STRONG_INLINE
int maxBlocks()
const {
return max_blocks_; }
382 inline bool ok()
const {
384 gpuError_t error = gpuStreamQuery(stream_->stream());
385 return (error == gpuSuccess) || (error == gpuErrorNotReady);
392 const StreamInterface* stream_;
398#define LAUNCH_GPU_KERNEL(kernel, gridsize, blocksize, sharedmem, device, ...) \
400 ::Eigen::internal::gpu_launch((kernel), dim3(gridsize), dim3(blocksize), (sharedmem), (device).stream(), \
407#include "../../../../Eigen/src/Core/util/GpuHipCudaUndefines.inc"
Namespace containing all symbols from the Eigen library.