Eigen-Contrib  5.0.1
 
Loading...
Searching...
No Matches
Eigen::gpu::DeviceMatrix< Scalar_ > Class Template Reference

#include <contrib/Eigen/src/GPU/DeviceMatrix.h>

Detailed Description

template<typename Scalar_>
class Eigen::gpu::DeviceMatrix< Scalar_ >

RAII wrapper for a dense column-major matrix in GPU device memory.

Template Parameters
Scalar_Element type: float, double, complex<float>, complex<double>

Owns a device allocation with tracked dimensions and leading dimension. An internal CUDA event records when the data was last written, enabling safe cross-stream consumption without user-visible synchronization.

Transfers come in synchronous and asynchronous variants: fromHost() / fromHostAsync() and toHost() / toHostAsync().

Public Member Functions

void addScaled (Context &ctx, Scalar alpha, const DeviceMatrix &x)
 
AdjointView< Scalar > adjoint () const
 
DeviceMatrix clone (cudaStream_t stream=nullptr) const
 
void copyFrom (Context &ctx, const DeviceMatrix &other)
 
void cwiseProduct (Context &ctx, const DeviceMatrix &a, const DeviceMatrix &b)
 
DeviceMatrix cwiseProduct (Context &ctx, const DeviceMatrix &other) const
 
Assignment< Scalar > device (Context &ctx)
 
 DeviceMatrix ()=default
 
 DeviceMatrix (const DeviceMatrix &other)
 
 DeviceMatrix (Index n)
 
 DeviceMatrix (Index rows, Index cols)
 
void divide (Context &ctx, Scalar alpha)
 
DeviceScalar< Scalar > dot (Context &ctx, const DeviceMatrix &other) const
 
void dot (Context &ctx, const DeviceMatrix &other, DeviceScalar< Scalar > &result) const
 
bool isView () const
 
LLTView< Scalar, Lower > llt () const
 
template<int UpLo>
LLTView< Scalar, UpLo > llt () const
 
LUView< Scalar > lu () const
 
DeviceMatrix & noalias ()
 
DeviceScalar< typename NumTraits< Scalar >::Real > norm (Context &ctx) const
 
DeviceMatrix & operator*= (const DeviceScalar< Scalar > &alpha)
 
DeviceMatrix & operator*= (Scalar alpha)
 
DeviceMatrix & operator+= (const DeviceMatrix &other)
 
DeviceMatrix & operator+= (const DeviceScaledDevice< Scalar > &expr)
 
DeviceMatrix & operator+= (const Scaled< DeviceMatrix > &expr)
 
DeviceMatrix & operator-= (const DeviceMatrix &other)
 
DeviceMatrix & operator-= (const DeviceScaledDevice< Scalar > &expr)
 
template<typename Lhs, typename Rhs>
DeviceMatrix & operator-= (const GemmExpr< Lhs, Rhs > &expr)
 
DeviceMatrix & operator-= (const Scaled< DeviceMatrix > &expr)
 
DeviceMatrix & operator/= (Scalar alpha)
 
DeviceMatrix & operator= (const DeviceAddExpr< Scalar > &expr)
 
DeviceMatrix & operator= (const Scaled< DeviceMatrix > &expr)
 
DeviceMatrix & operator= (const SpMVAffineExpr< Scalar > &expr)
 
DeviceMatrix & operator= (const SpMVExpr< Scalar > &expr)
 
void recordReady (cudaStream_t stream)
 
Scalar * release ()
 
void resize (Index rows, Index cols)
 
void scale (Context &ctx, Scalar alpha)
 
template<int UpLo>
SelfAdjointView< Scalar, UpLo > selfadjointView ()
 
template<int UpLo>
ConstSelfAdjointView< Scalar, UpLo > selfadjointView () const
 
void setZero (Context &ctx)
 
size_t sizeInBytes () const
 
DeviceScalar< typename NumTraits< Scalar >::Real > squaredNorm (Context &ctx) const
 
DeviceScalar< typename NumTraits< Scalar >::Real > stableNorm (Context &ctx) const
 
PlainMatrix toHost (cudaStream_t stream=nullptr) const
 
HostTransfer< Scalar > toHostAsync (cudaStream_t stream=nullptr) const
 
TransposeView< Scalar > transpose () const
 
template<int UpLo>
TriangularView< Scalar, UpLo > triangularView () const
 
void waitReady (cudaStream_t stream) const
 

Static Public Member Functions

static DeviceMatrix adopt (Scalar *device_ptr, Index rows, Index cols)
 
template<typename Derived>
static DeviceMatrix fromHost (const DenseBase< Derived > &host, cudaStream_t stream=nullptr)
 
static DeviceMatrix fromHostAsync (const Scalar *host_data, Index rows, Index cols, cudaStream_t stream)
 
static DeviceMatrix view (Scalar *device_ptr, Index rows, Index cols)
 

Constructor & Destructor Documentation

◆ DeviceMatrix() [1/4]

template<typename Scalar_>
Eigen::gpu::DeviceMatrix< Scalar_ >::DeviceMatrix ( )
default

Default: empty (0x0, no allocation).

◆ DeviceMatrix() [2/4]

template<typename Scalar_>
Eigen::gpu::DeviceMatrix< Scalar_ >::DeviceMatrix ( Index n)
inlineexplicit

Allocate an uninitialized column vector, mirroring Matrix<Scalar,Dynamic,1>(n) so generic solver code compiles unchanged.

◆ DeviceMatrix() [3/4]

template<typename Scalar_>
Eigen::gpu::DeviceMatrix< Scalar_ >::DeviceMatrix ( Index rows,
Index cols )
inline

Allocate uninitialized device memory for a rows x cols matrix.

◆ DeviceMatrix() [4/4]

template<typename Scalar_>
Eigen::gpu::DeviceMatrix< Scalar_ >::DeviceMatrix ( const DeviceMatrix< Scalar_ > & other)

Deep copy: a device-to-device cuBLAS copy on the thread-local Context, asynchronous and without a host transfer. Copies exist so that generic Eigen algorithm code with value semantics (p = precond.solve(residual) in internal::conjugate_gradient) compiles against DeviceMatrix; code that manages streams explicitly should prefer copyFrom(ctx, other). Defined out-of-line in DeviceDispatch.h, where Context is complete.

Member Function Documentation

◆ addScaled()

template<typename Scalar_>
void Eigen::gpu::DeviceMatrix< Scalar_ >::addScaled ( Context & ctx,
Scalar alpha,
const DeviceMatrix< Scalar_ > & x )

this += alpha * x (cuBLAS axpy). Requires same total size.

◆ adjoint()

template<typename Scalar_>
AdjointView< Scalar > Eigen::gpu::DeviceMatrix< Scalar_ >::adjoint ( ) const
inline

Adjoint view: maps to a GEMM operand with ConjTrans.

◆ adopt()

template<typename Scalar_>
static DeviceMatrix Eigen::gpu::DeviceMatrix< Scalar_ >::adopt ( Scalar * device_ptr,
Index rows,
Index cols )
inlinestatic

Adopt an existing device pointer. Caller relinquishes ownership.

◆ clone()

template<typename Scalar_>
DeviceMatrix Eigen::gpu::DeviceMatrix< Scalar_ >::clone ( cudaStream_t stream = nullptr) const
inline

Deep copy on device. Fully async — records an event on the result, no sync.

Parameters
streamCUDA stream for the D2D copy (default: stream 0).

◆ copyFrom()

template<typename Scalar_>
void Eigen::gpu::DeviceMatrix< Scalar_ >::copyFrom ( Context & ctx,
const DeviceMatrix< Scalar_ > & other )

Deep copy: this = other (cuBLAS copy). Resizes if needed.

◆ cwiseProduct() [1/2]

template<typename Scalar_>
void Eigen::gpu::DeviceMatrix< Scalar_ >::cwiseProduct ( Context & ctx,
const DeviceMatrix< Scalar_ > & a,
const DeviceMatrix< Scalar_ > & b )

In-place element-wise product: this[i] = a[i] * b[i]. Reuses this matrix's buffer when sizes match, avoiding cudaMalloc.

◆ cwiseProduct() [2/2]

template<typename Scalar_>
DeviceMatrix< Scalar_ > Eigen::gpu::DeviceMatrix< Scalar_ >::cwiseProduct ( Context & ctx,
const DeviceMatrix< Scalar_ > & other ) const

Element-wise product: result[i] = this[i] * other[i].

◆ device()

template<typename Scalar_>
Assignment< Scalar > Eigen::gpu::DeviceMatrix< Scalar_ >::device ( Context & ctx)
inline

Bind this matrix to a Context for expression assignment: d_C.device(ctx) = d_A * d_B;

◆ divide()

template<typename Scalar_>
void Eigen::gpu::DeviceMatrix< Scalar_ >::divide ( Context & ctx,
Scalar alpha )

this /= alpha. A true division for real Scalar (NPP divide-by-constant: one rounding per element, and no overflow of 1/alpha for subnormal alpha); complex Scalar scales by the host reciprocal, one extra rounding.

◆ dot() [1/2]

template<typename Scalar_>
DeviceScalar< typename DeviceMatrix< Scalar_ >::Scalar > Eigen::gpu::DeviceMatrix< Scalar_ >::dot ( Context & ctx,
const DeviceMatrix< Scalar_ > & other ) const

Dot product: this^H * other. The result stays on device until read through DeviceScalar's conversion to Scalar, which syncs.

◆ dot() [2/2]

template<typename Scalar_>
void Eigen::gpu::DeviceMatrix< Scalar_ >::dot ( Context & ctx,
const DeviceMatrix< Scalar_ > & other,
DeviceScalar< Scalar > & result ) const

The reductions above written into result's existing storage on ctx, instead of a new DeviceScalar: no allocation per call, so a loop or a captured CUDA graph repeats them without allocator traffic. result must live on ctx.stream().

◆ fromHost()

template<typename Scalar_>
template<typename Derived>
static DeviceMatrix Eigen::gpu::DeviceMatrix< Scalar_ >::fromHost ( const DenseBase< Derived > & host,
cudaStream_t stream = nullptr )
inlinestatic

Upload a host Eigen matrix to device memory (synchronous).

Copies to device via cudaMemcpyAsync on stream and synchronizes before returning. Plain contiguous column-major input is transferred directly; other expressions are first evaluated into a contiguous temporary.

Parameters
hostAny Eigen dense expression.
streamCUDA stream for the transfer (default: stream 0).

◆ fromHostAsync()

template<typename Scalar_>
static DeviceMatrix Eigen::gpu::DeviceMatrix< Scalar_ >::fromHostAsync ( const Scalar * host_data,
Index rows,
Index cols,
cudaStream_t stream )
inlinestatic

Upload from a raw host pointer to device memory (asynchronous).

Enqueues an async H2D copy on stream and records an internal event. The caller must keep host_data alive until the transfer completes.

Parameters
host_dataPointer to contiguous column-major host data.
rowsNumber of rows.
colsNumber of columns.
streamCUDA stream for the transfer.

◆ isView()

template<typename Scalar_>
bool Eigen::gpu::DeviceMatrix< Scalar_ >::isView ( ) const
inline

Whether this matrix holds storage borrowed through view(). Destroying a view does not free that storage, and release() on a view returns a pointer that its owner still frees. resize() to a different size replaces a view with storage of its own; resize() to the same size leaves it a view.

◆ llt() [1/2]

template<typename Scalar_>
LLTView< Scalar, Lower > Eigen::gpu::DeviceMatrix< Scalar_ >::llt ( ) const
inline

Cholesky view: d_A.llt().solve(d_B) → LltSolveExpr.

◆ llt() [2/2]

template<typename Scalar_>
template<int UpLo>
LLTView< Scalar, UpLo > Eigen::gpu::DeviceMatrix< Scalar_ >::llt ( ) const
inline

Cholesky view with explicit triangle: d_A.llt<Upper>().solve(d_B).

◆ lu()

template<typename Scalar_>
LUView< Scalar > Eigen::gpu::DeviceMatrix< Scalar_ >::lu ( ) const
inline

LU view: d_A.lu().solve(d_B) → LuSolveExpr.

◆ noalias()

template<typename Scalar_>
DeviceMatrix & Eigen::gpu::DeviceMatrix< Scalar_ >::noalias ( )
inline

No-op — every DeviceMatrix assignment is already implicitly noalias.

Eigen's Matrix falls back to a temporary when .noalias() is omitted, but DeviceMatrix dispatches straight to NVIDIA library calls, which offer no aliasing protection. For GEMM and SpMV the caller must therefore keep the operands clear of the destination; geam (d_C = d_A + alpha * d_B) is safe under aliasing. Debug asserts catch violations.

The method exists so tmp.noalias() = mat * p compiles for both types.

◆ norm()

template<typename Scalar_>
DeviceScalar< typename NumTraits< Scalar_ >::Real > Eigen::gpu::DeviceMatrix< Scalar_ >::norm ( Context & ctx) const

L2 norm as sqrt(squaredNorm()), like MatrixBase::norm(), without a host sync: unscaled, so it overflows once some |x_i| > sqrt(max) and loses accuracy when every |x_i| < sqrt(min).

◆ operator*=() [1/2]

template<typename Scalar_>
DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator*= ( const DeviceScalar< Scalar > & alpha)

this *= alpha (cuBLAS scal, device pointer mode). Avoids a host sync.

◆ operator*=() [2/2]

template<typename Scalar_>
DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator*= ( Scalar alpha)

this *= alpha (cuBLAS scal, host pointer mode).

◆ operator+=() [1/3]

template<typename Scalar_>
DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator+= ( const DeviceMatrix< Scalar_ > & other)

this += x (cuBLAS axpy with alpha=1).

◆ operator+=() [2/3]

template<typename Scalar_>
DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator+= ( const DeviceScaledDevice< Scalar > & expr)

this += DeviceScalar * x (cuBLAS axpy with POINTER_MODE_DEVICE).

◆ operator+=() [3/3]

template<typename Scalar_>
DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator+= ( const Scaled< DeviceMatrix< Scalar_ > > & expr)

this += alpha * x (cuBLAS axpy).

◆ operator-=() [1/4]

template<typename Scalar_>
DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator-= ( const DeviceMatrix< Scalar_ > & other)

this -= x (cuBLAS axpy with alpha=-1).

◆ operator-=() [2/4]

template<typename Scalar_>
DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator-= ( const DeviceScaledDevice< Scalar > & expr)

this -= DeviceScalar * x (cuBLAS axpy with negated device scalar).

◆ operator-=() [3/4]

template<typename Scalar_>
template<typename Lhs, typename Rhs>
DeviceMatrix & Eigen::gpu::DeviceMatrix< Scalar_ >::operator-= ( const GemmExpr< Lhs, Rhs > & expr)

Subtract a GEMM expression using the thread-local default Context.

◆ operator-=() [4/4]

template<typename Scalar_>
DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator-= ( const Scaled< DeviceMatrix< Scalar_ > > & expr)

this -= alpha * x (cuBLAS axpy with negated alpha).

◆ operator/=()

template<typename Scalar_>
DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator/= ( Scalar alpha)

this /= alpha, see divide().

◆ operator=() [1/4]

template<typename Scalar_>
DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator= ( const DeviceAddExpr< Scalar > & expr)

Assign from an add expression: d_C = alpha * d_A + beta * d_B (cuBLAS geam).

◆ operator=() [2/4]

template<typename Scalar_>
DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator= ( const Scaled< DeviceMatrix< Scalar_ > > & expr)

Assign from a scaled matrix: d_C = alpha * d_A (cuBLAS geam with beta=0). Also covers unary minus: d_C = -d_A. Safe when d_C aliases d_A.

◆ operator=() [3/4]

template<typename Scalar_>
DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator= ( const SpMVAffineExpr< Scalar > & expr)

Assign from an SpMV expression with a dense addend: d_y = d_b - d_A * d_x and the other sign combinations. Copies the addend into d_y (skipped when it is d_y), then one cuSPARSE call with beta = ±1. The addend must have the shape of the product, and d_x must not be d_y.

◆ operator=() [4/4]

template<typename Scalar_>
DeviceMatrix< Scalar_ > & Eigen::gpu::DeviceMatrix< Scalar_ >::operator= ( const SpMVExpr< Scalar > & expr)

Assign from an SpMV expression: d_y = d_A * d_x (one cuSPARSE call).

◆ recordReady()

template<typename Scalar_>
void Eigen::gpu::DeviceMatrix< Scalar_ >::recordReady ( cudaStream_t stream)
inline

Record that device data is ready after work on stream.

◆ release()

template<typename Scalar_>
Scalar * Eigen::gpu::DeviceMatrix< Scalar_ >::release ( )
inline

Give up the device pointer and zero the internal state. An owning matrix transfers ownership to the caller; a view (isView()) returns its borrowed pointer, which its owner still frees.

◆ resize()

template<typename Scalar_>
void Eigen::gpu::DeviceMatrix< Scalar_ >::resize ( Index rows,
Index cols )
inline

Discard contents and resize to (rows x cols). Contents are undefined afterwards. Keeps the existing allocation when it is large enough (capacity-aware: no cudaMalloc/cudaFree churn when cycling through same-or-smaller shapes); otherwise reallocates and clears the ready event.

◆ scale()

template<typename Scalar_>
void Eigen::gpu::DeviceMatrix< Scalar_ >::scale ( Context & ctx,
Scalar alpha )

this *= alpha (cuBLAS scal).

◆ selfadjointView() [1/2]

template<typename Scalar_>
template<int UpLo>
SelfAdjointView< Scalar, UpLo > Eigen::gpu::DeviceMatrix< Scalar_ >::selfadjointView ( )
inline

Self-adjoint view (mutable): d_C.selfadjointView<Lower>().rankUpdate(d_A).

◆ selfadjointView() [2/2]

template<typename Scalar_>
template<int UpLo>
ConstSelfAdjointView< Scalar, UpLo > Eigen::gpu::DeviceMatrix< Scalar_ >::selfadjointView ( ) const
inline

Self-adjoint view (const): d_A.selfadjointView<Lower>() * d_B → SymmExpr.

◆ setZero()

template<typename Scalar_>
void Eigen::gpu::DeviceMatrix< Scalar_ >::setZero ( Context & ctx)

Set all elements to zero.

◆ sizeInBytes()

template<typename Scalar_>
size_t Eigen::gpu::DeviceMatrix< Scalar_ >::sizeInBytes ( ) const
inline

Size of the device allocation in bytes.

◆ squaredNorm()

template<typename Scalar_>
DeviceScalar< typename NumTraits< Scalar_ >::Real > Eigen::gpu::DeviceMatrix< Scalar_ >::squaredNorm ( Context & ctx) const

Squared L2 norm via dot(x, x), without a host sync.

◆ stableNorm()

template<typename Scalar_>
DeviceScalar< typename NumTraits< Scalar_ >::Real > Eigen::gpu::DeviceMatrix< Scalar_ >::stableNorm ( Context & ctx) const

Overflow-safe L2 norm, like MatrixBase::stableNorm(): cuBLAS nrm2's scaled sum of squares, without a host sync. Eigen's iterative solver templates call it.

◆ toHost()

template<typename Scalar_>
PlainMatrix Eigen::gpu::DeviceMatrix< Scalar_ >::toHost ( cudaStream_t stream = nullptr) const
inline

Download device matrix to host memory (synchronous).

Waits on the internal ready event, enqueues a D2H copy on stream, synchronizes, and returns the host matrix directly.

Parameters
streamCUDA stream for the transfer (default: stream 0).

◆ toHostAsync()

template<typename Scalar_>
HostTransfer< Scalar > Eigen::gpu::DeviceMatrix< Scalar_ >::toHostAsync ( cudaStream_t stream = nullptr) const
inline

Enqueue an async device-to-host transfer and return a future.

Waits on the internal ready event (if any) to ensure the device data is valid, then enqueues the D2H copy on stream. Call HostTransfer::get() to block and retrieve the host matrix.

Parameters
streamCUDA stream for the transfer (default: stream 0).

◆ transpose()

template<typename Scalar_>
TransposeView< Scalar > Eigen::gpu::DeviceMatrix< Scalar_ >::transpose ( ) const
inline

Transpose view: maps to a GEMM operand with Trans.

◆ triangularView()

template<typename Scalar_>
template<int UpLo>
TriangularView< Scalar, UpLo > Eigen::gpu::DeviceMatrix< Scalar_ >::triangularView ( ) const
inline

Triangular view: d_A.triangularView<Lower>().solve(d_B) → TrsmExpr.

◆ view()

template<typename Scalar_>
static DeviceMatrix Eigen::gpu::DeviceMatrix< Scalar_ >::view ( Scalar * device_ptr,
Index rows,
Index cols )
inlinestatic

Construct a non-owning view over an existing device pointer.

The pointer is borrowed: destruction does not free, and the underlying storage must outlive this view. This chains decomposition outputs (e.g. svd.d_matrixU()) into downstream cuBLAS expressions without an intervening D2D copy, and supports the full read interface. Do not assign through a view: the borrowed pointer would be silently replaced, leaving the owner intact.

◆ waitReady()

template<typename Scalar_>
void Eigen::gpu::DeviceMatrix< Scalar_ >::waitReady ( cudaStream_t stream) const
inline

Make stream wait until the device data is ready. No-op if no event recorded, or if the consumer stream is the same as the producer stream (CUDA guarantees in-order execution within a stream).


The documentation for this class was generated from the following files: