Eigen-Contrib  5.0.1
 
Loading...
Searching...
No Matches
TensorDeviceGpu.h
1// This file is part of Eigen, a lightweight C++ template library
2// for linear algebra.
3//
4// Copyright (C) 2014 Benoit Steiner <benoit.steiner.goog@gmail.com>
5//
6// This Source Code Form is subject to the terms of the Mozilla
7// Public License v. 2.0. If a copy of the MPL was not distributed
8// with this file, You can obtain one at http://mozilla.org/MPL/2.0/.
9// SPDX-License-Identifier: MPL-2.0
10
11#if defined(EIGEN_USE_GPU) && !defined(EIGEN_TENSOR_TENSOR_DEVICE_GPU_H)
12#define EIGEN_TENSOR_TENSOR_DEVICE_GPU_H
13
14// IWYU pragma: private
15#include "./InternalHeaderCheck.h"
16
17#include "../../../../Eigen/src/Core/util/GpuHipCudaDefines.inc"
18#include "../../../../Eigen/src/Core/util/GpuRuntime.h"
19
20namespace Eigen {
21
22static const int kGpuScratchSize = 1024;
23
24// The device facts Eigen consults, read one at a time. The reason is portability, not speed: the opt-in
25// shared-memory limit and memory-pool support are not fields a gpuDeviceProp_t carries under both backends, while
26// the attribute enumerators exist in each. (Measured on CUDA 13.3, a warm gpuGetDeviceProperties costs 0.3-0.5 us
27// against 1.4-1.5 us for these nine queries, and a cold call of either is dominated by runtime initialization.)
28struct GpuDeviceAttributes {
29 int multiProcessorCount;
30 int maxThreadsPerBlock;
31 int maxThreadsPerMultiProcessor;
32 int sharedMemPerBlock;
33 int sharedMemPerBlockOptin;
34 int computeCapabilityMajor;
35 int computeCapabilityMinor;
36 int warpSize;
37 int memoryPoolsSupported;
38};
39
40// Filled once for every visible device at first use, which is thread-safe by the initialization rules and needs no
41// lock. A few microseconds per device, so there is nothing to gain from filling it per device on demand.
42// Templated on the attribute enum: CUDA spells it cudaDeviceAttr and HIP hipDeviceAttribute_t, and the .inc pair
43// aliases the enumerators but not the type.
44template <typename GpuDeviceAttr>
45inline int GetGpuDeviceAttribute(GpuDeviceAttr attribute, int device) {
46#if !defined(EIGEN_USE_HIP)
47 if (attribute == gpuDevAttrMemoryPoolsSupported) {
48 int driver_version = 0;
49 EIGEN_GPU_RUNTIME_CHECK(cudaDriverGetVersion(&driver_version));
50 // CUDA minor-version compatibility permits drivers that predate this attribute.
51 if (driver_version < 11020) return 0;
52 }
53#endif
54 int value = 0;
55 const gpuError_t error = gpuDeviceGetAttribute(&value, attribute, device);
56#if defined(EIGEN_USE_HIP)
57 // HIP 5.x declares the opt-in enumerator but does not implement the query.
58 if (attribute == gpuDevAttrMaxSharedMemoryPerBlockOptin && error == hipErrorInvalidValue) {
59 (void)hipGetLastError();
60 EIGEN_GPU_RUNTIME_CHECK(gpuDeviceGetAttribute(&value, gpuDevAttrMaxSharedMemoryPerBlock, device));
61 return value;
62 }
63#endif
64 EIGEN_GPU_RUNTIME_CHECK(error);
65 return value;
66}
67
68inline const std::vector<GpuDeviceAttributes>& GetGpuDeviceAttributes() {
69 // The device count is dynamic; keep the cache alive for GpuDevice users in static destructors.
70 static const std::vector<GpuDeviceAttributes>* kAttributes = [] {
71 int num_devices = 0;
72 EIGEN_GPU_RUNTIME_CHECK(gpuGetDeviceCount(&num_devices));
73 auto* attributes = new std::vector<GpuDeviceAttributes>(num_devices);
74 for (int device = 0; device < num_devices; ++device) {
75 (*attributes)[device] = {
76 /*multiProcessorCount=*/GetGpuDeviceAttribute(gpuDevAttrMultiProcessorCount, device),
77 /*maxThreadsPerBlock=*/GetGpuDeviceAttribute(gpuDevAttrMaxThreadsPerBlock, device),
78 /*maxThreadsPerMultiProcessor=*/GetGpuDeviceAttribute(gpuDevAttrMaxThreadsPerMultiProcessor, device),
79 /*sharedMemPerBlock=*/GetGpuDeviceAttribute(gpuDevAttrMaxSharedMemoryPerBlock, device),
80 /*sharedMemPerBlockOptin=*/GetGpuDeviceAttribute(gpuDevAttrMaxSharedMemoryPerBlockOptin, device),
81 /*computeCapabilityMajor=*/GetGpuDeviceAttribute(gpuDevAttrComputeCapabilityMajor, device),
82 /*computeCapabilityMinor=*/GetGpuDeviceAttribute(gpuDevAttrComputeCapabilityMinor, device),
83 /*warpSize=*/GetGpuDeviceAttribute(gpuDevAttrWarpSize, device),
84 /*memoryPoolsSupported=*/GetGpuDeviceAttribute(gpuDevAttrMemoryPoolsSupported, device),
85 };
86 }
87 return attributes;
88 }();
89 return *kAttributes;
90}
91
92inline const GpuDeviceAttributes& GetGpuDeviceAttributes(int device) {
93 const std::vector<GpuDeviceAttributes>& attributes = GetGpuDeviceAttributes();
94 eigen_assert(device >= 0 && device < static_cast<int>(attributes.size()) && "no such GPU device");
95 return attributes[device];
96}
97
98// Attributes for StreamInterface implementations that follow the calling thread's current device.
99// GpuStreamDevice instead queries the device that owns its stream.
100inline const GpuDeviceAttributes& GetCurrentGpuDeviceAttributes() {
101 int device = 0;
102 EIGEN_GPU_RUNTIME_CHECK(gpuGetDevice(&device));
103 return GetGpuDeviceAttributes(device);
104}
105
106// This defines an interface that GPUDevice can take to use
107// HIP / CUDA streams underneath.
108class StreamInterface {
109 public:
110 virtual ~StreamInterface() = default;
111
112 virtual const gpuStream_t& stream() const = 0;
113 virtual const gpuDeviceProp_t& deviceProperties() const = 0;
114
115 // The attributes of the device this interface's stream runs on. The default reports the device the calling
116 // thread is bound to, which is what an implementation that follows the current device wants; one that owns a
117 // device index overrides it, as GpuStreamDevice does, so that the answer follows the stream rather than
118 // whatever device the caller happens to have selected.
119 virtual const GpuDeviceAttributes& deviceAttributes() const { return GetCurrentGpuDeviceAttributes(); }
120
121 // Allocate memory on the actual device where the computation will run
122 virtual void* allocate(size_t num_bytes) const = 0;
123 virtual void deallocate(void* buffer) const = 0;
124
125 // Return a scratchpad buffer of size 1k
126 virtual void* scratchpad() const = 0;
127
128 // Return a semaphore. The semaphore is initially initialized to 0, and
129 // each kernel using it is responsible for resetting to 0 upon completion
130 // to maintain the invariant that the semaphore is always equal to 0 upon
131 // each kernel start.
132 virtual unsigned int* semaphore() const = 0;
133};
134
135class GpuDeviceProperties {
136 public:
137 static const GpuDeviceProperties& instance() {
138 static const GpuDeviceProperties* kInstance = new GpuDeviceProperties();
139
140 return *kInstance;
141 }
142
143 EIGEN_STRONG_INLINE const gpuDeviceProp_t& get(int device) const {
144 eigen_assert(device >= 0 && device < static_cast<int>(device_properties_.size()) && "no such GPU device");
145 return device_properties_[device];
146 }
147
148 private:
149 GpuDeviceProperties() = default;
150
151 static std::vector<gpuDeviceProp_t> GetDeviceProperties() {
152 int num_devices = 0;
153 EIGEN_GPU_RUNTIME_CHECK(gpuGetDeviceCount(&num_devices));
154 std::vector<gpuDeviceProp_t> device_properties(num_devices);
155 for (int i = 0; i < num_devices; ++i) {
156 EIGEN_GPU_RUNTIME_CHECK(gpuGetDeviceProperties(&device_properties[i], i));
157 }
158
159 return device_properties;
160 }
161
162 std::vector<gpuDeviceProp_t> device_properties_ = GetDeviceProperties();
163};
164
165EIGEN_ALWAYS_INLINE const GpuDeviceProperties& GetGpuDeviceProperties() { return GpuDeviceProperties::instance(); }
166
167EIGEN_ALWAYS_INLINE const gpuDeviceProp_t& GetGpuDeviceProperties(int device) {
168 return GetGpuDeviceProperties().get(device);
169}
170
171static const gpuStream_t default_stream = gpuStreamDefault;
172
173class GpuStreamDevice : public StreamInterface {
174 public:
175 // Use the default stream on the current device
176 GpuStreamDevice() : stream_(&default_stream), scratch_(nullptr), semaphore_(nullptr) {
177 EIGEN_GPU_RUNTIME_CHECK(gpuGetDevice(&device_));
178 }
179 // Use the default stream on the specified device
180 GpuStreamDevice(int device) : stream_(&default_stream), device_(device), scratch_(nullptr), semaphore_(nullptr) {}
181 // Use the specified stream. Note that it's the
182 // caller's responsibility to ensure that the stream can run on
183 // the specified device. If no device is specified the code
184 // assumes that the stream is associated to the current gpu device.
185 GpuStreamDevice(const gpuStream_t* stream, int device = -1)
186 : stream_(stream), device_(device), scratch_(nullptr), semaphore_(nullptr) {
187 if (device < 0) {
188 EIGEN_GPU_RUNTIME_CHECK(gpuGetDevice(&device_));
189 } else {
190 int num_devices = 0;
191 EIGEN_GPU_RUNTIME_CHECK(gpuGetDeviceCount(&num_devices));
192 EIGEN_UNUSED_VARIABLE(num_devices);
193 gpu_assert(device < num_devices);
194 device_ = device;
195 }
196 }
197
198 ~GpuStreamDevice() override {
199 if (scratch_) {
200 deallocate(scratch_);
201 }
202 }
203
204 const gpuStream_t& stream() const override { return *stream_; }
205 const gpuDeviceProp_t& deviceProperties() const override { return GetGpuDeviceProperties(device_); }
206 const GpuDeviceAttributes& deviceAttributes() const override { return GetGpuDeviceAttributes(device_); }
207 void* allocate(size_t num_bytes) const override {
208 EIGEN_GPU_RUNTIME_CHECK(gpuSetDevice(device_));
209 void* result = nullptr;
210 EIGEN_GPU_RUNTIME_CHECK(gpuMalloc(&result, num_bytes));
211 gpu_assert(result != nullptr);
212 return result;
213 }
214 void deallocate(void* buffer) const override {
215 EIGEN_GPU_RUNTIME_CHECK(gpuSetDevice(device_));
216 gpu_assert(buffer != nullptr);
217 EIGEN_GPU_RUNTIME_CHECK(gpuFree(buffer));
218 }
219
220 void* scratchpad() const override {
221 if (scratch_ == nullptr) {
222 scratch_ = allocate(kGpuScratchSize + sizeof(unsigned int));
223 }
224 return scratch_;
225 }
226
227 unsigned int* semaphore() const override {
228 if (semaphore_ == nullptr) {
229 char* scratch = static_cast<char*>(scratchpad()) + kGpuScratchSize;
230 semaphore_ = reinterpret_cast<unsigned int*>(scratch);
231 EIGEN_GPU_RUNTIME_CHECK(gpuMemsetAsync(semaphore_, 0, sizeof(unsigned int), *stream_));
232 }
233 return semaphore_;
234 }
235
236 private:
237 const gpuStream_t* stream_;
238 int device_;
239 mutable void* scratch_;
240 mutable unsigned int* semaphore_;
241};
242
243struct GpuDevice {
244 // The StreamInterface is not owned: the caller is
245 // responsible for its initialization and eventual destruction.
246 explicit GpuDevice(const StreamInterface* stream) : stream_(stream), max_blocks_(INT_MAX) { eigen_assert(stream); }
247 // Nothing reads max_blocks_ any more; the executor sizes its grid from the device's own limits.
248 EIGEN_DEPRECATED explicit GpuDevice(const StreamInterface* stream, int num_blocks)
249 : stream_(stream), max_blocks_(num_blocks) {
250 eigen_assert(stream);
251 }
252 // TODO(bsteiner): This is an internal API, we should not expose it.
253 EIGEN_STRONG_INLINE const gpuStream_t& stream() const { return stream_->stream(); }
254
255 EIGEN_STRONG_INLINE void* allocate(size_t num_bytes) const { return stream_->allocate(num_bytes); }
256
257 EIGEN_STRONG_INLINE void deallocate(void* buffer) const { stream_->deallocate(buffer); }
258
259 EIGEN_STRONG_INLINE void* allocate_temp(size_t num_bytes) const { return stream_->allocate(num_bytes); }
260
261 EIGEN_STRONG_INLINE void deallocate_temp(void* buffer) const { stream_->deallocate(buffer); }
262
263 template <typename Type>
264 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Type get(Type data) const {
265 return data;
266 }
267
268 EIGEN_STRONG_INLINE void* scratchpad() const { return stream_->scratchpad(); }
269
270 EIGEN_STRONG_INLINE unsigned int* semaphore() const { return stream_->semaphore(); }
271
272 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void memcpy(void* dst, const void* src, size_t n) const {
273#ifndef EIGEN_GPU_COMPILE_PHASE
274 EIGEN_GPU_RUNTIME_CHECK(gpuMemcpyAsync(dst, src, n, gpuMemcpyDeviceToDevice, stream_->stream()));
275#else
276 EIGEN_UNUSED_VARIABLE(dst);
277 EIGEN_UNUSED_VARIABLE(src);
278 EIGEN_UNUSED_VARIABLE(n);
279 eigen_assert(false && "The default device should be used instead to generate kernel code");
280#endif
281 }
282
283 EIGEN_STRONG_INLINE void memcpyHostToDevice(void* dst, const void* src, size_t n) const {
284 EIGEN_GPU_RUNTIME_CHECK(gpuMemcpyAsync(dst, src, n, gpuMemcpyHostToDevice, stream_->stream()));
285 }
286
287 EIGEN_STRONG_INLINE void memcpyDeviceToHost(void* dst, const void* src, size_t n) const {
288 EIGEN_GPU_RUNTIME_CHECK(gpuMemcpyAsync(dst, src, n, gpuMemcpyDeviceToHost, stream_->stream()));
289 }
290
291 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void memset(void* buffer, int c, size_t n) const {
292#ifndef EIGEN_GPU_COMPILE_PHASE
293 EIGEN_GPU_RUNTIME_CHECK(gpuMemsetAsync(buffer, c, n, stream_->stream()));
294#else
295 EIGEN_UNUSED_VARIABLE(buffer);
296 EIGEN_UNUSED_VARIABLE(c);
297 EIGEN_UNUSED_VARIABLE(n);
298 eigen_assert(false && "The default device should be used instead to generate kernel code");
299#endif
300 }
301
302 template <typename T>
303 EIGEN_STRONG_INLINE void fill(T* begin, T* end, const T& value) const {
304#ifndef EIGEN_GPU_COMPILE_PHASE
305 const size_t count = end - begin;
306 // Split value into bytes and run memset with stride.
307 const int value_size = sizeof(value);
308 char* buffer = (char*)begin;
309 char* value_bytes = (char*)(&value);
310 // If all value bytes are equal, then a single memset can be much faster.
311 bool use_single_memset = true;
312 for (int i = 1; i < value_size; ++i) {
313 if (value_bytes[i] != value_bytes[0]) {
314 use_single_memset = false;
315 }
316 }
317
318 if (use_single_memset) {
319 EIGEN_GPU_RUNTIME_CHECK(gpuMemsetAsync(buffer, value_bytes[0], count * sizeof(T), stream_->stream()));
320 } else {
321 for (int b = 0; b < value_size; ++b) {
322 EIGEN_GPU_RUNTIME_CHECK(gpuMemset2DAsync(buffer + b, value_size, value_bytes[b], 1, count, stream_->stream()));
323 }
324 }
325#else
326 EIGEN_UNUSED_VARIABLE(begin);
327 EIGEN_UNUSED_VARIABLE(end);
328 EIGEN_UNUSED_VARIABLE(value);
329 eigen_assert(false && "The default device should be used instead to generate kernel code");
330#endif
331 }
332
333 // The warp (wavefront) width, which is what "a thread" means to the block-size heuristics that ask.
334 EIGEN_STRONG_INLINE size_t numThreads() const { return static_cast<size_t>(warpSize()); }
335
336 EIGEN_STRONG_INLINE size_t firstLevelCacheSize() const {
337 // FIXME: Return a more accurate cache size.
338 return 48 * 1024;
339 }
340
341 EIGEN_STRONG_INLINE size_t lastLevelCacheSize() const {
342 // We won't try to take advantage of the l2 cache for the time being, and
343 // there is no l3 cache on hip/cuda devices.
344 return firstLevelCacheSize();
345 }
346
347 EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void synchronize() const {
348#ifndef EIGEN_GPU_COMPILE_PHASE
349 EIGEN_GPU_RUNTIME_CHECK(gpuStreamSynchronize(stream_->stream()));
350#else
351 gpu_assert(false && "The default device should be used instead to generate kernel code");
352#endif
353 }
354
355 EIGEN_STRONG_INLINE int getNumGpuMultiProcessors() const { return stream_->deviceProperties().multiProcessorCount; }
356 EIGEN_STRONG_INLINE int maxGpuThreadsPerBlock() const { return stream_->deviceProperties().maxThreadsPerBlock; }
357 EIGEN_STRONG_INLINE int maxGpuThreadsPerMultiProcessor() const {
358 return stream_->deviceProperties().maxThreadsPerMultiProcessor;
359 }
360 EIGEN_STRONG_INLINE int sharedMemPerBlock() const {
361 return static_cast<int>(stream_->deviceProperties().sharedMemPerBlock);
362 }
363 EIGEN_STRONG_INLINE int majorDeviceVersion() const { return stream_->deviceProperties().major; }
364 EIGEN_STRONG_INLINE int minorDeviceVersion() const { return stream_->deviceProperties().minor; }
365
366 // Read from the StreamInterface's attributes rather than from its properties: gpuDeviceProp_t does not carry
367 // the opt-in shared-memory limit or memory-pool support portably. Like the properties above, they describe the
368 // device the stream and its allocations belong to, which need not be the device the calling thread is bound to.
369 EIGEN_STRONG_INLINE int warpSize() const { return stream_->deviceAttributes().warpSize; }
370 // The shared memory a kernel may request with gpuFuncSetAttribute, or the ordinary block limit when HIP does
371 // not implement the opt-in attribute.
372 EIGEN_STRONG_INLINE int sharedMemPerBlockOptin() const { return stream_->deviceAttributes().sharedMemPerBlockOptin; }
373 // Whether gpuMallocAsync and its pool are available on this device and driver.
374 EIGEN_STRONG_INLINE bool memoryPoolsSupported() const {
375 return stream_->deviceAttributes().memoryPoolsSupported != 0;
376 }
377
378 EIGEN_DEPRECATED EIGEN_STRONG_INLINE int maxBlocks() const { return max_blocks_; }
379
380 // This function checks if the GPU runtime recorded an error for the
381 // underlying stream device.
382 inline bool ok() const {
383#ifdef EIGEN_GPUCC
384 gpuError_t error = gpuStreamQuery(stream_->stream());
385 return (error == gpuSuccess) || (error == gpuErrorNotReady);
386#else
387 return false;
388#endif
389 }
390
391 private:
392 const StreamInterface* stream_;
393 int max_blocks_;
394};
395
396// Launches `kernel` on the device's stream through internal::gpu_launch (GpuRuntime.h), which reports a failed
397// launch through EIGEN_GPU_RUNTIME_CHECK. A single statement, so it composes with an unbraced `if`.
398#define LAUNCH_GPU_KERNEL(kernel, gridsize, blocksize, sharedmem, device, ...) \
399 do { \
400 ::Eigen::internal::gpu_launch((kernel), dim3(gridsize), dim3(blocksize), (sharedmem), (device).stream(), \
401 __VA_ARGS__); \
402 } while (0)
403
404} // end namespace Eigen
405
406// undefine all the gpu* macros we defined at the beginning of the file
407#include "../../../../Eigen/src/Core/util/GpuHipCudaUndefines.inc"
408
409#endif // EIGEN_TENSOR_TENSOR_DEVICE_GPU_H
Namespace containing all symbols from the Eigen library.