Skip to main content

CudaBackend

Struct CudaBackend 

Source
pub struct CudaBackend { /* private fields */ }

Implementations§

Source§

impl CudaBackend

Source

pub fn new(device_id: CudaDeviceId) -> Result<Self, CudaDeviceError>

Create a new CubeCL backend for the caller-selected CUDA device.

§Examples
use tenferro_gpu::{cuda::CudaBackend, cuda::CudaDeviceError, cuda::CudaDeviceId};

let _ctor: fn(CudaDeviceId) -> Result<CudaBackend, CudaDeviceError> = CudaBackend::new;
§Errors

Returns CudaDeviceError::Discovery when device discovery fails, CudaDeviceError::Unavailable when the selected device is not discovered, or CudaDeviceError::Initialization when CUDA runtime, context, or CubeCL client initialization fails.

Source

pub fn runtime(&self) -> &CudaRuntime

Borrow the underlying CubeCL runtime.

§Examples
use tenferro_gpu::{cuda::CudaBackend, cuda::CudaRuntime};

let _runtime: fn(&CudaBackend) -> &CudaRuntime = CudaBackend::runtime;
Source

pub fn device_id(&self) -> CudaDeviceId

Return the caller-selected CUDA device identity used by this backend.

§Examples
use tenferro_gpu::{cuda::CudaBackend, cuda::CudaDeviceId};

let _device_id: fn(&CudaBackend) -> CudaDeviceId = CudaBackend::device_id;
Source

pub fn runtime_identity(&self) -> CudaRuntimeIdentity

Return the opaque identity of this exact executable backend instance.

Clones of a backend return the same identity. Independently constructed backends return different identities even when they target the same CUDA device ordinal.

§Examples
use tenferro_gpu::cuda::CudaBackend;

let _identity = CudaBackend::runtime_identity;
Source

pub fn clear_cuda_extension_cache(&self) -> Result<()>

Clear CUDA extension-owned backend state.

§Errors

Returns crate::Error::RuntimeState if the extension cache mutex is poisoned.

Source

pub fn cuda_extension_cache_stats(&self) -> Result<CacheStats>

Return CUDA extension cache stats.

§Errors

Returns crate::Error::RuntimeState if the extension cache mutex is poisoned.

Source

pub fn cuda_extension_cache_max_entries(&self) -> Result<NonZeroUsize>

Return the CUDA extension cache entry bound.

§Errors

Returns crate::Error::RuntimeState if the extension cache mutex is poisoned.

Source

pub fn cuda_extension_cache_max_retained_bytes(&self) -> Result<NonZeroUsize>

Return the CUDA extension cache logical retained-byte bound.

§Errors

Returns crate::Error::RuntimeState if the extension cache mutex is poisoned.

Source

pub fn set_cuda_extension_cache_max_entries( &self, max_entries: NonZeroUsize, ) -> Result<()>

Configure the CUDA extension cache entry bound.

§Errors

Returns crate::Error::RuntimeState if the extension cache mutex is poisoned while changing the bound.

Source

pub fn set_cuda_extension_cache_max_retained_bytes( &self, max_retained_bytes: NonZeroUsize, ) -> Result<()>

Configure the CUDA extension cache logical retained-byte bound.

§Errors

Returns crate::Error::RuntimeState if the extension cache mutex is poisoned while changing the bound.

Source

pub fn cutensor_plan_cache_stats(&self) -> Result<CacheStats>

Return cuTENSOR contraction plan cache stats.

The returned entry count is the number of retained cuTENSOR contraction plans inside the CUDA backend’s extension cache entry. Logical retained bytes cover plan metadata, not the shared per-stream device scratch; CudaBackend::cutensor_workspace_stats reports that separately. The cache byte limit is therefore not a total device-memory limit.

§Errors

Returns crate::Error::RuntimeState if the cache mutex is poisoned.

Source

pub fn cutensor_workspace_stats(&self) -> Result<CutensorWorkspaceStats>

Return the retained shared cuTENSOR contraction scratch, in bytes.

All cached cuTENSOR contraction plans share one lazily grown workspace per physical stream slot, so this is the sum over slots of the capacity each slot currently holds. It is bounded by CudaBackend::set_cutensor_workspace_max_retained_bytes, reported separately from the extension-cache byte statistics, and released by clearing the extension cache or dropping the backend. This is retained scratch, not total device memory: a workspace in use by a queued contraction, a retiring allocation, and vendor-internal memory are all excluded.

Read this to size a retention cap: it is the high-water demand of the workload shapes that have run so far.

§Examples
use tenferro_gpu::cuda::{cuda_devices, gpu_available, CudaBackend};

// `gpu_available` never panics without a CUDA driver, so this
// example also runs in CPU-only doctest environments.
if gpu_available() {
    let device = cuda_devices()?.remove(0);
    let backend = CudaBackend::new(device.id())?;
    let stats = backend.cutensor_workspace_stats()?;
    println!("{:?}", (stats.retained_entries, stats.retained_bytes));
}
§Errors

Returns crate::Error::RuntimeState if the cache mutex is poisoned.

Source

pub fn cutensor_workspace_bytes(&self) -> Result<u64>

Return the device bytes retained by the shared cuTENSOR contraction scratch. Equal to CudaBackend::cutensor_workspace_stats().retained_bytes.

§Examples
use tenferro_gpu::cuda::{cuda_devices, gpu_available, CudaBackend};

// `gpu_available` never panics without a CUDA driver, so this
// example also runs in CPU-only doctest environments.
if gpu_available() {
    let device = cuda_devices()?.remove(0);
    let backend = CudaBackend::new(device.id())?;
    println!("{} bytes retained", backend.cutensor_workspace_bytes()?);
}
§Errors

Returns crate::Error::RuntimeState if the cache mutex is poisoned.

Source

pub fn cutensor_workspace_max_retained_bytes(&self) -> u64

Return the configured retention cap for shared cuTENSOR contraction scratch, in bytes.

The default is 10 GiB. See CudaBackend::set_cutensor_workspace_max_retained_bytes for the contract; this value is not a device-memory reservation. The cap is plain backend state, so reading it cannot fail and never creates cache state.

§Examples
use tenferro_gpu::cuda::{cuda_devices, gpu_available, CudaBackend};

// `gpu_available` never panics without a CUDA driver, so this
// example also runs in CPU-only doctest environments.
if gpu_available() {
    let device = cuda_devices()?.remove(0);
    let backend = CudaBackend::new(device.id())?;
    println!("cap {} bytes", backend.cutensor_workspace_max_retained_bytes());
}
Source

pub fn set_cutensor_workspace_max_retained_bytes( &self, bytes: u64, ) -> Result<()>

Configure the retention cap for shared cuTENSOR contraction scratch.

The cap bounds how much scratch the backend keeps for reuse, summed over physical stream slots. It never refuses a contraction: a requirement that does not fit the remaining cap runs in a temporary workspace that is released afterwards, and shrinking the cap drops retained buffers without evicting any cached plan. Other slots are never evicted to make room.

0 disables retention entirely; it is not “unlimited”. The default (10 GiB) is finite but is not a practical memory protection, and neither the cap nor the reported statistics bound total device memory.

Setting a cap below the steady-state working set makes matching contractions allocate and retire their scratch on every call, which can increase workspace-retirement stream barrier fallbacks. To choose a value, run the workload and read the CudaBackend::cutensor_workspace_bytes high-water: retaining every slot’s rounded high-water needs a cap of at least the sum of next_power_of_two(max(request, 1 MiB)) over the stream slots.

The setting is stored on the backend, so it survives CudaBackend::clear_cuda_extension_cache and extension-cache eviction, and is shared by clones of this backend.

§Examples
use tenferro_gpu::cuda::{cuda_devices, gpu_available, CudaBackend};

// `gpu_available` never panics without a CUDA driver, so this
// example also runs in CPU-only doctest environments.
if gpu_available() {
    let device = cuda_devices()?.remove(0);
    let backend = CudaBackend::new(device.id())?;
    backend.set_cutensor_workspace_max_retained_bytes(4 << 30)?;
}
§Errors

Returns crate::Error::RuntimeState if the plan-cache mutex is poisoned while releasing retained buffers.

Source

pub fn cutensor_workspace_temporary_uses(&self) -> u64

Return how many contractions ran in a temporary shared-scratch workspace because their requirement did not fit the retention cap.

This is the direct signal that the cap is binding. A nonzero value means the matching contractions allocated and retired their scratch on every call instead of reusing a retained buffer; the high-water from CudaBackend::cutensor_workspace_bytes then under-reports the real requirement. Raise CudaBackend::set_cutensor_workspace_max_retained_bytes until this stops increasing, or accept the churn deliberately.

The count is cumulative for the backend, shared by clones, and is not reset by CudaBackend::clear_cuda_extension_cache; diff two reads to measure an interval. Reading it cannot fail and never creates cache state.

§Examples
use tenferro_gpu::cuda::{cuda_devices, gpu_available, CudaBackend};

// `gpu_available` never panics without a CUDA driver, so this
// example also runs in CPU-only doctest environments.
if gpu_available() {
    let device = cuda_devices()?.remove(0);
    let backend = CudaBackend::new(device.id())?;
    println!("{} temporary uses", backend.cutensor_workspace_temporary_uses());
}
Source

pub fn cutensor_workspace_retirement_stats( &self, ) -> Result<WorkspaceRetirementStats>

Return deferred cuTENSOR workspace retirement counters.

Retirements are deferred until the workspace’s stream reaches the event recorded at retirement time. in_flight is the number of workspaces whose handle has not returned to the CubeCL pool yet.

§Errors

Returns crate::Error::RuntimeState if the retirement queue lock is poisoned.

Source

pub fn cutensor_plan_cache_max_entries(&self) -> Result<NonZeroUsize>

Return the cuTENSOR contraction plan entry bound.

§Errors

Returns crate::Error::RuntimeState if the cache mutex is poisoned.

Source

pub fn set_cutensor_plan_cache_max_entries( &self, max_entries: NonZeroUsize, ) -> Result<()>

Configure the cuTENSOR contraction plan entry bound.

The cache is initialized if it does not already exist so a setting made before the first CUDA dot_general call is preserved.

§Errors

Returns crate::Error::RuntimeState if the cache mutex is poisoned.

Source

pub fn cutensor_permutation_plan_cache_stats(&self) -> Result<CacheStats>

Return cuTENSOR structural permutation plan cache stats.

The returned entry count is the number of retained cuTENSOR permutation plans inside the CUDA backend’s extension cache entry. Logical retained bytes include cached descriptor and plan state.

§Errors

Returns crate::Error::RuntimeState if the cache mutex is poisoned.

Source

pub fn cutensor_permutation_plan_cache_max_entries( &self, ) -> Result<NonZeroUsize>

Return the cuTENSOR structural permutation plan entry bound.

§Errors

Returns crate::Error::RuntimeState if the cache mutex is poisoned.

Source

pub fn set_cutensor_permutation_plan_cache_max_entries( &self, max_entries: NonZeroUsize, ) -> Result<()>

Configure the cuTENSOR structural permutation plan entry bound.

The cache is initialized if it does not already exist so a setting made before the first CUDA structural permutation call is preserved.

§Errors

Returns crate::Error::RuntimeState if the cache mutex is poisoned.

Trait Implementations§

Source§

impl BackendRuntimeCache for CudaBackend

Source§

impl BackendSessionHost for CudaBackend

Source§

fn with_backend_session<R: Send>( &mut self, f: impl FnOnce(&mut dyn BackendSession) -> R + Send, ) -> Result<R, SessionEntryError>

Open one backend session and run f inside it. Read more
Source§

impl Clone for CudaBackend

Source§

fn clone(&self) -> CudaBackend

Returns a duplicate of the value. Read more
1.0.0 (const: unstable) · Source§

fn clone_from(&mut self, source: &Self)

Performs copy-assignment from source. Read more
Source§

impl Debug for CudaBackend

Source§

fn fmt(&self, f: &mut Formatter<'_>) -> Result

Formats the value using the given formatter. Read more
Source§

impl DotGeneralPreparation for CudaBackend

Source§

fn prepare( &self, request: DotGeneralPrepareRequest<'_>, ) -> Result<PrepareCapability, PrepareError>

Prepare one dot-general operation. Read more
Source§

impl ElementwiseRuntime for CudaBackend

Source§

fn prepare( &self, request: ElementwisePrepareRequest<'_>, ) -> Result<PrepareCapability, PrepareError>

Prepare one elementwise operation. Read more
Source§

fn max_fused_region_inputs(&self) -> Option<usize>

Largest number of distinct external inputs one fused elementwise region may read on this engine, or None when the engine has no such limit. Read more
Source§

impl IndexingRuntime for CudaBackend

Source§

fn prepare( &self, request: IndexingPrepareRequest<'_>, ) -> Result<PrepareCapability, PrepareError>

Prepare one indexing operation. Read more
Source§

impl LayoutRuntime for CudaBackend

Source§

fn prepare( &self, request: LayoutPrepareRequest<'_>, ) -> Result<PrepareCapability, PrepareError>

Prepare one layout operation. Read more
Source§

impl ReductionRuntime for CudaBackend

Source§

fn prepare( &self, request: ReductionPrepareRequest<'_>, ) -> Result<PrepareCapability, PrepareError>

Prepare one reduction operation. Read more
Source§

impl RuntimeCacheOwner for CudaBackend

Source§

fn cache_stats(&self) -> Result<CacheStats, CacheOwnerError>

Return this owner’s current cache statistics. Read more
Source§

fn clear_caches(&self) -> Result<(), CacheOwnerError>

Clear this owner’s retained caches. Read more
Source§

impl TensorBackend for CudaBackend

Source§

impl TensorBackendCapability for CudaBackend

Source§

fn backend_id(&self) -> BackendId

Source§

fn capabilities(&self) -> &'static [OperationCapability]

Source§

fn capability(&self, query: CapabilityQuery) -> Option<OperationCapability>

Look up one operation/dtype capability for this backend. Read more
Source§

fn require_capability( &self, query: CapabilityQuery, axis: CapabilityAxis, ) -> Result<OperationCapability, Error>

Require support for one operation/dtype/axis, returning a structured unsupported error otherwise. Read more
Source§

impl TensorDeviceTransfer for CudaBackend

Source§

fn download_to_host(&mut self, tensor: TensorRead<'_>) -> Result<Tensor>

Explicitly copy a provider-owned read target into host storage. Read more
Source§

fn upload_host_tensor(&mut self, tensor: TensorRead<'_>) -> Result<Tensor>

Explicitly copy a host read target into provider storage. Read more

Auto Trait Implementations§

Blanket Implementations§

Source§

impl<T> Any for T
where T: 'static + ?Sized,

Source§

fn type_id(&self) -> TypeId

Gets the TypeId of self. Read more
Source§

impl<T> Borrow<T> for T
where T: ?Sized,

Source§

fn borrow(&self) -> &T

Immutably borrows from an owned value. Read more
Source§

impl<T> BorrowMut<T> for T
where T: ?Sized,

Source§

fn borrow_mut(&mut self) -> &mut T

Mutably borrows from an owned value. Read more
§

impl<T> ByRef<T> for T

§

fn by_ref(&self) -> &T

§

impl<ST, DT> CastableFrom<ST, Initialized, Initialized> for DT
where ST: ?Sized, DT: ?Sized,

§

impl<ST, DT> CastableFrom<ST, Uninit, Uninit> for DT
where ST: ?Sized, DT: ?Sized,

§

impl<C> CloneExpand for C
where C: Clone,

§

fn __expand_clone_method(&self, _scope: &mut Scope) -> C

Source§

impl<T> CloneToUninit for T
where T: Clone,

Source§

unsafe fn clone_to_uninit(&self, dest: *mut u8)

🔬This is a nightly-only experimental API. (clone_to_uninit)
Performs copy-assignment from self to dest. Read more
§

impl<T> Downcast<T> for T

§

fn downcast(&self) -> &T

Source§

impl<T> From<T> for T

Source§

fn from(t: T) -> T

Returns the argument unchanged.

§

impl<T, U> Imply<T> for U
where T: ?Sized, U: ?Sized,

Source§

impl<T, U> Into<U> for T
where U: From<T>,

Source§

fn into(self) -> U

Calls U::from(self).

That is, this conversion is whatever the implementation of From<T> for U chooses to do.

§

impl<T> IntoComptime for T

§

fn comptime(self) -> Self

Source§

impl<T> IntoEither for T

Source§

fn into_either(self, into_left: bool) -> Either<Self, Self> ⓘ

Converts self into a Left variant of Either<Self, Self> if into_left is true. Converts self into a Right variant of Either<Self, Self> otherwise. Read more
Source§

fn into_either_with<F>(self, into_left: F) -> Either<Self, Self> ⓘ
where F: FnOnce(&Self) -> bool,

Converts self into a Left variant of Either<Self, Self> if into_left(&self) returns true. Converts self into a Right variant of Either<Self, Self> otherwise. Read more
§

impl<T> MaybeSend for T
where T: Send,

§

impl<T> MaybeSendSync for T
where T: Send + Sync,

§

impl<T> MaybeSync for T
where T: Sync,

§

impl<T> Pointable for T

§

const ALIGN: usize

The alignment of pointer.
§

type Init = T

The type for initializers.
§

unsafe fn init(init: <T as Pointable>::Init) -> usize

Initializes a with the given initializer. Read more
§

unsafe fn deref<'a>(ptr: usize) -> &'a T

Dereferences the given pointer. Read more
§

unsafe fn deref_mut<'a>(ptr: usize) -> &'a mut T

Mutably dereferences the given pointer. Read more
§

unsafe fn drop(ptr: usize)

Drops the object pointed to by the given pointer. Read more
§

impl<T> Read<Exclusive, BecauseExclusive> for T
where T: ?Sized,

Source§

impl<T> ToOwned for T
where T: Clone,

Source§

type Owned = T

The resulting type after obtaining ownership.
Source§

fn to_owned(&self) -> T

Creates owned data from borrowed data, usually by cloning. Read more
Source§

fn clone_into(&self, target: &mut T)

Uses borrowed data to replace owned data, usually by cloning. Read more
Source§

impl<T, U> TryFrom<U> for T
where U: Into<T>,

Source§

type Error = Infallible

The type returned in the event of a conversion error.
Source§

fn try_from(value: U) -> Result<T, <T as TryFrom<U>>::Error>

Performs the conversion.
Source§

impl<T, U> TryInto<U> for T
where U: TryFrom<T>,

Source§

type Error = <U as TryFrom<T>>::Error

The type returned in the event of a conversion error.
Source§

fn try_into(self) -> Result<U, <U as TryFrom<T>>::Error>

Performs the conversion.
§

impl<T> TuneInputs for T
where T: Clone + Send + Sync + 'static,

§

type At<'a> = T

The concrete input type at lifetime 'a.
§

impl<T> Upcast<T> for T

§

fn upcast(&self) -> Option<&T>

§

impl<T> WasmNotSend for T
where T: Send,

§

impl<T> WasmNotSendSync for T
where T: WasmNotSend + WasmNotSync,

§

impl<T> WasmNotSync for T
where T: Sync,