![]() |
Eigen-Contrib
5.0.1
|
#include <contrib/Eigen/src/GPU/DeviceMatrix.h>
RAII wrapper for a dense column-major matrix in GPU device memory.
| Scalar_ | Element type: float, double, complex<float>, complex<double> |
Owns a device allocation with tracked dimensions and leading dimension. An internal CUDA event records when the data was last written, enabling safe cross-stream consumption without user-visible synchronization.
Transfers come in synchronous and asynchronous variants: fromHost() / fromHostAsync() and toHost() / toHostAsync().
Public Member Functions | |
| void | addScaled (Context &ctx, Scalar alpha, const DeviceMatrix &x) |
| AdjointView< Scalar > | adjoint () const |
| DeviceMatrix | clone (cudaStream_t stream=nullptr) const |
| void | copyFrom (Context &ctx, const DeviceMatrix &other) |
| void | cwiseProduct (Context &ctx, const DeviceMatrix &a, const DeviceMatrix &b) |
| DeviceMatrix | cwiseProduct (Context &ctx, const DeviceMatrix &other) const |
| Assignment< Scalar > | device (Context &ctx) |
| DeviceMatrix ()=default | |
| DeviceMatrix (const DeviceMatrix &other) | |
| DeviceMatrix (Index n) | |
| DeviceMatrix (Index rows, Index cols) | |
| void | divide (Context &ctx, Scalar alpha) |
| DeviceScalar< Scalar > | dot (Context &ctx, const DeviceMatrix &other) const |
| void | dot (Context &ctx, const DeviceMatrix &other, DeviceScalar< Scalar > &result) const |
| bool | isView () const |
| LLTView< Scalar, Lower > | llt () const |
| template<int UpLo> | |
| LLTView< Scalar, UpLo > | llt () const |
| LUView< Scalar > | lu () const |
| DeviceMatrix & | noalias () |
| DeviceScalar< typename NumTraits< Scalar >::Real > | norm (Context &ctx) const |
| DeviceMatrix & | operator*= (const DeviceScalar< Scalar > &alpha) |
| DeviceMatrix & | operator*= (Scalar alpha) |
| DeviceMatrix & | operator+= (const DeviceMatrix &other) |
| DeviceMatrix & | operator+= (const DeviceScaledDevice< Scalar > &expr) |
| DeviceMatrix & | operator+= (const Scaled< DeviceMatrix > &expr) |
| DeviceMatrix & | operator-= (const DeviceMatrix &other) |
| DeviceMatrix & | operator-= (const DeviceScaledDevice< Scalar > &expr) |
| template<typename Lhs, typename Rhs> | |
| DeviceMatrix & | operator-= (const GemmExpr< Lhs, Rhs > &expr) |
| DeviceMatrix & | operator-= (const Scaled< DeviceMatrix > &expr) |
| DeviceMatrix & | operator/= (Scalar alpha) |
| DeviceMatrix & | operator= (const DeviceAddExpr< Scalar > &expr) |
| DeviceMatrix & | operator= (const Scaled< DeviceMatrix > &expr) |
| DeviceMatrix & | operator= (const SpMVAffineExpr< Scalar > &expr) |
| DeviceMatrix & | operator= (const SpMVExpr< Scalar > &expr) |
| void | recordReady (cudaStream_t stream) |
| Scalar * | release () |
| void | resize (Index rows, Index cols) |
| void | scale (Context &ctx, Scalar alpha) |
| template<int UpLo> | |
| SelfAdjointView< Scalar, UpLo > | selfadjointView () |
| template<int UpLo> | |
| ConstSelfAdjointView< Scalar, UpLo > | selfadjointView () const |
| void | setZero (Context &ctx) |
| size_t | sizeInBytes () const |
| DeviceScalar< typename NumTraits< Scalar >::Real > | squaredNorm (Context &ctx) const |
| DeviceScalar< typename NumTraits< Scalar >::Real > | stableNorm (Context &ctx) const |
| PlainMatrix | toHost (cudaStream_t stream=nullptr) const |
| HostTransfer< Scalar > | toHostAsync (cudaStream_t stream=nullptr) const |
| TransposeView< Scalar > | transpose () const |
| template<int UpLo> | |
| TriangularView< Scalar, UpLo > | triangularView () const |
| void | waitReady (cudaStream_t stream) const |
Static Public Member Functions | |
| static DeviceMatrix | adopt (Scalar *device_ptr, Index rows, Index cols) |
| template<typename Derived> | |
| static DeviceMatrix | fromHost (const DenseBase< Derived > &host, cudaStream_t stream=nullptr) |
| static DeviceMatrix | fromHostAsync (const Scalar *host_data, Index rows, Index cols, cudaStream_t stream) |
| static DeviceMatrix | view (Scalar *device_ptr, Index rows, Index cols) |
|
default |
Default: empty (0x0, no allocation).
|
inlineexplicit |
Allocate an uninitialized column vector, mirroring Matrix<Scalar,Dynamic,1>(n) so generic solver code compiles unchanged.
|
inline |
Allocate uninitialized device memory for a rows x cols matrix.
| Eigen::gpu::DeviceMatrix< Scalar_ >::DeviceMatrix | ( | const DeviceMatrix< Scalar_ > & | other | ) |
Deep copy: a device-to-device cuBLAS copy on the thread-local Context, asynchronous and without a host transfer. Copies exist so that generic Eigen algorithm code with value semantics (p = precond.solve(residual) in internal::conjugate_gradient) compiles against DeviceMatrix; code that manages streams explicitly should prefer copyFrom(ctx, other). Defined out-of-line in DeviceDispatch.h, where Context is complete.
| void Eigen::gpu::DeviceMatrix< Scalar_ >::addScaled | ( | Context & | ctx, |
| Scalar | alpha, | ||
| const DeviceMatrix< Scalar_ > & | x ) |
this += alpha * x (cuBLAS axpy). Requires same total size.
|
inline |
Adjoint view: maps to a GEMM operand with ConjTrans.
|
inlinestatic |
Adopt an existing device pointer. Caller relinquishes ownership.
|
inline |
Deep copy on device. Fully async — records an event on the result, no sync.
| stream | CUDA stream for the D2D copy (default: stream 0). |
| void Eigen::gpu::DeviceMatrix< Scalar_ >::copyFrom | ( | Context & | ctx, |
| const DeviceMatrix< Scalar_ > & | other ) |
Deep copy: this = other (cuBLAS copy). Resizes if needed.
| void Eigen::gpu::DeviceMatrix< Scalar_ >::cwiseProduct | ( | Context & | ctx, |
| const DeviceMatrix< Scalar_ > & | a, | ||
| const DeviceMatrix< Scalar_ > & | b ) |
In-place element-wise product: this[i] = a[i] * b[i]. Reuses this matrix's buffer when sizes match, avoiding cudaMalloc.
| DeviceMatrix< Scalar_ > Eigen::gpu::DeviceMatrix< Scalar_ >::cwiseProduct | ( | Context & | ctx, |
| const DeviceMatrix< Scalar_ > & | other ) const |
Element-wise product: result[i] = this[i] * other[i].
|
inline |
Bind this matrix to a Context for expression assignment: d_C.device(ctx) = d_A * d_B;
| void Eigen::gpu::DeviceMatrix< Scalar_ >::divide | ( | Context & | ctx, |
| Scalar | alpha ) |
this /= alpha. A true division for real Scalar (NPP divide-by-constant: one rounding per element, and no overflow of 1/alpha for subnormal alpha); complex Scalar scales by the host reciprocal, one extra rounding.
| DeviceScalar< typename DeviceMatrix< Scalar_ >::Scalar > Eigen::gpu::DeviceMatrix< Scalar_ >::dot | ( | Context & | ctx, |
| const DeviceMatrix< Scalar_ > & | other ) const |
Dot product: this^H * other. The result stays on device until read through DeviceScalar's conversion to Scalar, which syncs.
| void Eigen::gpu::DeviceMatrix< Scalar_ >::dot | ( | Context & | ctx, |
| const DeviceMatrix< Scalar_ > & | other, | ||
| DeviceScalar< Scalar > & | result ) const |
The reductions above written into result's existing storage on ctx, instead of a new DeviceScalar: no allocation per call, so a loop or a captured CUDA graph repeats them without allocator traffic. result must live on ctx.stream().
|
inlinestatic |
Upload a host Eigen matrix to device memory (synchronous).
Copies to device via cudaMemcpyAsync on stream and synchronizes before returning. Plain contiguous column-major input is transferred directly; other expressions are first evaluated into a contiguous temporary.
| host | Any Eigen dense expression. |
| stream | CUDA stream for the transfer (default: stream 0). |
|
inlinestatic |
Upload from a raw host pointer to device memory (asynchronous).
Enqueues an async H2D copy on stream and records an internal event. The caller must keep host_data alive until the transfer completes.
| host_data | Pointer to contiguous column-major host data. |
| rows | Number of rows. |
| cols | Number of columns. |
| stream | CUDA stream for the transfer. |
|
inline |
|
inline |
Cholesky view: d_A.llt().solve(d_B) → LltSolveExpr.
|
inline |
Cholesky view with explicit triangle: d_A.llt<Upper>().solve(d_B).
|
inline |
LU view: d_A.lu().solve(d_B) → LuSolveExpr.
|
inline |
No-op — every DeviceMatrix assignment is already implicitly noalias.
Eigen's Matrix falls back to a temporary when .noalias() is omitted, but DeviceMatrix dispatches straight to NVIDIA library calls, which offer no aliasing protection. For GEMM and SpMV the caller must therefore keep the operands clear of the destination; geam (d_C = d_A + alpha * d_B) is safe under aliasing. Debug asserts catch violations.
The method exists so tmp.noalias() = mat * p compiles for both types.
| DeviceScalar< typename NumTraits< Scalar_ >::Real > Eigen::gpu::DeviceMatrix< Scalar_ >::norm | ( | Context & | ctx | ) | const |
L2 norm as sqrt(squaredNorm()), like MatrixBase::norm(), without a host sync: unscaled, so it overflows once some |x_i| > sqrt(max) and loses accuracy when every |x_i| < sqrt(min).
| DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator*= | ( | const DeviceScalar< Scalar > & | alpha | ) |
this *= alpha (cuBLAS scal, device pointer mode). Avoids a host sync.
| DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator*= | ( | Scalar | alpha | ) |
this *= alpha (cuBLAS scal, host pointer mode).
| DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator+= | ( | const DeviceMatrix< Scalar_ > & | other | ) |
this += x (cuBLAS axpy with alpha=1).
| DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator+= | ( | const DeviceScaledDevice< Scalar > & | expr | ) |
this += DeviceScalar * x (cuBLAS axpy with POINTER_MODE_DEVICE).
| DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator+= | ( | const Scaled< DeviceMatrix< Scalar_ > > & | expr | ) |
this += alpha * x (cuBLAS axpy).
| DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator-= | ( | const DeviceMatrix< Scalar_ > & | other | ) |
this -= x (cuBLAS axpy with alpha=-1).
| DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator-= | ( | const DeviceScaledDevice< Scalar > & | expr | ) |
this -= DeviceScalar * x (cuBLAS axpy with negated device scalar).
| DeviceMatrix & Eigen::gpu::DeviceMatrix< Scalar_ >::operator-= | ( | const GemmExpr< Lhs, Rhs > & | expr | ) |
Subtract a GEMM expression using the thread-local default Context.
| DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator-= | ( | const Scaled< DeviceMatrix< Scalar_ > > & | expr | ) |
this -= alpha * x (cuBLAS axpy with negated alpha).
| DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator/= | ( | Scalar | alpha | ) |
this /= alpha, see divide().
| DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator= | ( | const DeviceAddExpr< Scalar > & | expr | ) |
Assign from an add expression: d_C = alpha * d_A + beta * d_B (cuBLAS geam).
| DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator= | ( | const Scaled< DeviceMatrix< Scalar_ > > & | expr | ) |
Assign from a scaled matrix: d_C = alpha * d_A (cuBLAS geam with beta=0). Also covers unary minus: d_C = -d_A. Safe when d_C aliases d_A.
| DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator= | ( | const SpMVAffineExpr< Scalar > & | expr | ) |
Assign from an SpMV expression with a dense addend: d_y = d_b - d_A * d_x and the other sign combinations. Copies the addend into d_y (skipped when it is d_y), then one cuSPARSE call with beta = ±1. The addend must have the shape of the product, and d_x must not be d_y.
| DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator= | ( | const SpMVExpr< Scalar > & | expr | ) |
Assign from an SpMV expression: d_y = d_A * d_x (one cuSPARSE call).
|
inline |
Record that device data is ready after work on stream.
|
inline |
Give up the device pointer and zero the internal state. An owning matrix transfers ownership to the caller; a view (isView()) returns its borrowed pointer, which its owner still frees.
|
inline |
Discard contents and resize to (rows x cols). Contents are undefined afterwards. Keeps the existing allocation when it is large enough (capacity-aware: no cudaMalloc/cudaFree churn when cycling through same-or-smaller shapes); otherwise reallocates and clears the ready event.
| void Eigen::gpu::DeviceMatrix< Scalar_ >::scale | ( | Context & | ctx, |
| Scalar | alpha ) |
this *= alpha (cuBLAS scal).
|
inline |
Self-adjoint view (mutable): d_C.selfadjointView<Lower>().rankUpdate(d_A).
|
inline |
Self-adjoint view (const): d_A.selfadjointView<Lower>() * d_B → SymmExpr.
| void Eigen::gpu::DeviceMatrix< Scalar_ >::setZero | ( | Context & | ctx | ) |
Set all elements to zero.
|
inline |
Size of the device allocation in bytes.
| DeviceScalar< typename NumTraits< Scalar_ >::Real > Eigen::gpu::DeviceMatrix< Scalar_ >::squaredNorm | ( | Context & | ctx | ) | const |
Squared L2 norm via dot(x, x), without a host sync.
| DeviceScalar< typename NumTraits< Scalar_ >::Real > Eigen::gpu::DeviceMatrix< Scalar_ >::stableNorm | ( | Context & | ctx | ) | const |
Overflow-safe L2 norm, like MatrixBase::stableNorm(): cuBLAS nrm2's scaled sum of squares, without a host sync. Eigen's iterative solver templates call it.
|
inline |
Download device matrix to host memory (synchronous).
Waits on the internal ready event, enqueues a D2H copy on stream, synchronizes, and returns the host matrix directly.
| stream | CUDA stream for the transfer (default: stream 0). |
|
inline |
Enqueue an async device-to-host transfer and return a future.
Waits on the internal ready event (if any) to ensure the device data is valid, then enqueues the D2H copy on stream. Call HostTransfer::get() to block and retrieve the host matrix.
| stream | CUDA stream for the transfer (default: stream 0). |
|
inline |
Transpose view: maps to a GEMM operand with Trans.
|
inline |
Triangular view: d_A.triangularView<Lower>().solve(d_B) → TrsmExpr.
|
inlinestatic |
Construct a non-owning view over an existing device pointer.
The pointer is borrowed: destruction does not free, and the underlying storage must outlive this view. This chains decomposition outputs (e.g. svd.d_matrixU()) into downstream cuBLAS expressions without an intervening D2D copy, and supports the full read interface. Do not assign through a view: the borrowed pointer would be silently replaced, leaving the owner intact.
|
inline |
Make stream wait until the device data is ready. No-op if no event recorded, or if the consumer stream is the same as the producer stream (CUDA guarantees in-order execution within a stream).