pub struct CudaBackend { /* private fields */ }Implementations§
Source§impl CudaBackend
impl CudaBackend
Sourcepub fn new(device_id: CudaDeviceId) -> Result<Self, CudaDeviceError>
pub fn new(device_id: CudaDeviceId) -> Result<Self, CudaDeviceError>
Create a new CubeCL backend for the caller-selected CUDA device.
§Examples
use tenferro_gpu::{cuda::CudaBackend, cuda::CudaDeviceError, cuda::CudaDeviceId};
let _ctor: fn(CudaDeviceId) -> Result<CudaBackend, CudaDeviceError> = CudaBackend::new;§Errors
Returns CudaDeviceError::Discovery when device discovery fails,
CudaDeviceError::Unavailable when the selected device is not
discovered, or CudaDeviceError::Initialization when CUDA runtime,
context, or CubeCL client initialization fails.
Sourcepub fn runtime(&self) -> &CudaRuntime
pub fn runtime(&self) -> &CudaRuntime
Borrow the underlying CubeCL runtime.
§Examples
use tenferro_gpu::{cuda::CudaBackend, cuda::CudaRuntime};
let _runtime: fn(&CudaBackend) -> &CudaRuntime = CudaBackend::runtime;Sourcepub fn device_id(&self) -> CudaDeviceId
pub fn device_id(&self) -> CudaDeviceId
Return the caller-selected CUDA device identity used by this backend.
§Examples
use tenferro_gpu::{cuda::CudaBackend, cuda::CudaDeviceId};
let _device_id: fn(&CudaBackend) -> CudaDeviceId = CudaBackend::device_id;Sourcepub fn runtime_identity(&self) -> CudaRuntimeIdentity
pub fn runtime_identity(&self) -> CudaRuntimeIdentity
Return the opaque identity of this exact executable backend instance.
Clones of a backend return the same identity. Independently constructed backends return different identities even when they target the same CUDA device ordinal.
§Examples
use tenferro_gpu::cuda::CudaBackend;
let _identity = CudaBackend::runtime_identity;Sourcepub fn clear_cuda_extension_cache(&self) -> Result<()>
pub fn clear_cuda_extension_cache(&self) -> Result<()>
Clear CUDA extension-owned backend state.
§Errors
Returns crate::Error::RuntimeState if the extension cache mutex is
poisoned.
Sourcepub fn cuda_extension_cache_stats(&self) -> Result<CacheStats>
pub fn cuda_extension_cache_stats(&self) -> Result<CacheStats>
Return CUDA extension cache stats.
§Errors
Returns crate::Error::RuntimeState if the extension cache mutex is
poisoned.
Sourcepub fn cuda_extension_cache_max_entries(&self) -> Result<NonZeroUsize>
pub fn cuda_extension_cache_max_entries(&self) -> Result<NonZeroUsize>
Return the CUDA extension cache entry bound.
§Errors
Returns crate::Error::RuntimeState if the extension cache mutex is
poisoned.
Sourcepub fn cuda_extension_cache_max_retained_bytes(&self) -> Result<NonZeroUsize>
pub fn cuda_extension_cache_max_retained_bytes(&self) -> Result<NonZeroUsize>
Return the CUDA extension cache logical retained-byte bound.
§Errors
Returns crate::Error::RuntimeState if the extension cache mutex is
poisoned.
Sourcepub fn set_cuda_extension_cache_max_entries(
&self,
max_entries: NonZeroUsize,
) -> Result<()>
pub fn set_cuda_extension_cache_max_entries( &self, max_entries: NonZeroUsize, ) -> Result<()>
Configure the CUDA extension cache entry bound.
§Errors
Returns crate::Error::RuntimeState if the extension cache mutex is
poisoned while changing the bound.
Sourcepub fn set_cuda_extension_cache_max_retained_bytes(
&self,
max_retained_bytes: NonZeroUsize,
) -> Result<()>
pub fn set_cuda_extension_cache_max_retained_bytes( &self, max_retained_bytes: NonZeroUsize, ) -> Result<()>
Configure the CUDA extension cache logical retained-byte bound.
§Errors
Returns crate::Error::RuntimeState if the extension cache mutex is
poisoned while changing the bound.
Sourcepub fn cutensor_plan_cache_stats(&self) -> Result<CacheStats>
pub fn cutensor_plan_cache_stats(&self) -> Result<CacheStats>
Return cuTENSOR contraction plan cache stats.
The returned entry count is the number of retained cuTENSOR contraction
plans inside the CUDA backend’s extension cache entry. Logical retained
bytes cover plan metadata, not the shared per-stream device scratch;
CudaBackend::cutensor_workspace_stats reports that separately. The
cache byte limit is therefore not a total device-memory limit.
§Errors
Returns crate::Error::RuntimeState if the cache mutex is poisoned.
Sourcepub fn cutensor_workspace_stats(&self) -> Result<CutensorWorkspaceStats>
pub fn cutensor_workspace_stats(&self) -> Result<CutensorWorkspaceStats>
Return the retained shared cuTENSOR contraction scratch, in bytes.
All cached cuTENSOR contraction plans share one lazily grown workspace
per physical stream slot, so this is the sum over slots of the capacity
each slot currently holds. It is bounded by
CudaBackend::set_cutensor_workspace_max_retained_bytes, reported
separately from the extension-cache byte statistics, and released by
clearing the extension cache or dropping the backend. This is retained
scratch, not total device memory: a workspace in use by a queued
contraction, a retiring allocation, and vendor-internal memory are all
excluded.
Read this to size a retention cap: it is the high-water demand of the workload shapes that have run so far.
§Examples
use tenferro_gpu::cuda::{cuda_devices, gpu_available, CudaBackend};
// `gpu_available` never panics without a CUDA driver, so this
// example also runs in CPU-only doctest environments.
if gpu_available() {
let device = cuda_devices()?.remove(0);
let backend = CudaBackend::new(device.id())?;
let stats = backend.cutensor_workspace_stats()?;
println!("{:?}", (stats.retained_entries, stats.retained_bytes));
}§Errors
Returns crate::Error::RuntimeState if the cache mutex is poisoned.
Sourcepub fn cutensor_workspace_bytes(&self) -> Result<u64>
pub fn cutensor_workspace_bytes(&self) -> Result<u64>
Return the device bytes retained by the shared cuTENSOR contraction
scratch. Equal to
CudaBackend::cutensor_workspace_stats().retained_bytes.
§Examples
use tenferro_gpu::cuda::{cuda_devices, gpu_available, CudaBackend};
// `gpu_available` never panics without a CUDA driver, so this
// example also runs in CPU-only doctest environments.
if gpu_available() {
let device = cuda_devices()?.remove(0);
let backend = CudaBackend::new(device.id())?;
println!("{} bytes retained", backend.cutensor_workspace_bytes()?);
}§Errors
Returns crate::Error::RuntimeState if the cache mutex is poisoned.
Sourcepub fn cutensor_workspace_max_retained_bytes(&self) -> u64
pub fn cutensor_workspace_max_retained_bytes(&self) -> u64
Return the configured retention cap for shared cuTENSOR contraction scratch, in bytes.
The default is 10 GiB. See
CudaBackend::set_cutensor_workspace_max_retained_bytes for the
contract; this value is not a device-memory reservation. The cap is
plain backend state, so reading it cannot fail and never creates cache
state.
§Examples
use tenferro_gpu::cuda::{cuda_devices, gpu_available, CudaBackend};
// `gpu_available` never panics without a CUDA driver, so this
// example also runs in CPU-only doctest environments.
if gpu_available() {
let device = cuda_devices()?.remove(0);
let backend = CudaBackend::new(device.id())?;
println!("cap {} bytes", backend.cutensor_workspace_max_retained_bytes());
}Sourcepub fn set_cutensor_workspace_max_retained_bytes(
&self,
bytes: u64,
) -> Result<()>
pub fn set_cutensor_workspace_max_retained_bytes( &self, bytes: u64, ) -> Result<()>
Configure the retention cap for shared cuTENSOR contraction scratch.
The cap bounds how much scratch the backend keeps for reuse, summed over physical stream slots. It never refuses a contraction: a requirement that does not fit the remaining cap runs in a temporary workspace that is released afterwards, and shrinking the cap drops retained buffers without evicting any cached plan. Other slots are never evicted to make room.
0 disables retention entirely; it is not “unlimited”. The default
(10 GiB) is finite but is not a practical memory protection, and neither
the cap nor the reported statistics bound total device memory.
Setting a cap below the steady-state working set makes matching
contractions allocate and retire their scratch on every call, which can
increase workspace-retirement stream barrier fallbacks. To choose a
value, run the workload and read the
CudaBackend::cutensor_workspace_bytes high-water: retaining every
slot’s rounded high-water needs a cap of at least the sum of
next_power_of_two(max(request, 1 MiB)) over the stream slots.
The setting is stored on the backend, so it survives
CudaBackend::clear_cuda_extension_cache and extension-cache eviction,
and is shared by clones of this backend.
§Examples
use tenferro_gpu::cuda::{cuda_devices, gpu_available, CudaBackend};
// `gpu_available` never panics without a CUDA driver, so this
// example also runs in CPU-only doctest environments.
if gpu_available() {
let device = cuda_devices()?.remove(0);
let backend = CudaBackend::new(device.id())?;
backend.set_cutensor_workspace_max_retained_bytes(4 << 30)?;
}§Errors
Returns crate::Error::RuntimeState if the plan-cache mutex is
poisoned while releasing retained buffers.
Sourcepub fn cutensor_workspace_temporary_uses(&self) -> u64
pub fn cutensor_workspace_temporary_uses(&self) -> u64
Return how many contractions ran in a temporary shared-scratch workspace because their requirement did not fit the retention cap.
This is the direct signal that the cap is binding. A nonzero value means
the matching contractions allocated and retired their scratch on every
call instead of reusing a retained buffer; the high-water from
CudaBackend::cutensor_workspace_bytes then under-reports the real
requirement. Raise
CudaBackend::set_cutensor_workspace_max_retained_bytes until this
stops increasing, or accept the churn deliberately.
The count is cumulative for the backend, shared by clones, and is not
reset by CudaBackend::clear_cuda_extension_cache; diff two reads to
measure an interval. Reading it cannot fail and never creates cache
state.
§Examples
use tenferro_gpu::cuda::{cuda_devices, gpu_available, CudaBackend};
// `gpu_available` never panics without a CUDA driver, so this
// example also runs in CPU-only doctest environments.
if gpu_available() {
let device = cuda_devices()?.remove(0);
let backend = CudaBackend::new(device.id())?;
println!("{} temporary uses", backend.cutensor_workspace_temporary_uses());
}Sourcepub fn cutensor_workspace_retirement_stats(
&self,
) -> Result<WorkspaceRetirementStats>
pub fn cutensor_workspace_retirement_stats( &self, ) -> Result<WorkspaceRetirementStats>
Return deferred cuTENSOR workspace retirement counters.
Retirements are deferred until the workspace’s stream reaches the event
recorded at retirement time. in_flight is the number of workspaces
whose handle has not returned to the CubeCL pool yet.
§Errors
Returns crate::Error::RuntimeState if the retirement queue lock is
poisoned.
Sourcepub fn cutensor_plan_cache_max_entries(&self) -> Result<NonZeroUsize>
pub fn cutensor_plan_cache_max_entries(&self) -> Result<NonZeroUsize>
Return the cuTENSOR contraction plan entry bound.
§Errors
Returns crate::Error::RuntimeState if the cache mutex is poisoned.
Sourcepub fn set_cutensor_plan_cache_max_entries(
&self,
max_entries: NonZeroUsize,
) -> Result<()>
pub fn set_cutensor_plan_cache_max_entries( &self, max_entries: NonZeroUsize, ) -> Result<()>
Configure the cuTENSOR contraction plan entry bound.
The cache is initialized if it does not already exist so a setting made
before the first CUDA dot_general call is preserved.
§Errors
Returns crate::Error::RuntimeState if the cache mutex is poisoned.
Sourcepub fn cutensor_permutation_plan_cache_stats(&self) -> Result<CacheStats>
pub fn cutensor_permutation_plan_cache_stats(&self) -> Result<CacheStats>
Return cuTENSOR structural permutation plan cache stats.
The returned entry count is the number of retained cuTENSOR permutation plans inside the CUDA backend’s extension cache entry. Logical retained bytes include cached descriptor and plan state.
§Errors
Returns crate::Error::RuntimeState if the cache mutex is poisoned.
Sourcepub fn cutensor_permutation_plan_cache_max_entries(
&self,
) -> Result<NonZeroUsize>
pub fn cutensor_permutation_plan_cache_max_entries( &self, ) -> Result<NonZeroUsize>
Return the cuTENSOR structural permutation plan entry bound.
§Errors
Returns crate::Error::RuntimeState if the cache mutex is poisoned.
Sourcepub fn set_cutensor_permutation_plan_cache_max_entries(
&self,
max_entries: NonZeroUsize,
) -> Result<()>
pub fn set_cutensor_permutation_plan_cache_max_entries( &self, max_entries: NonZeroUsize, ) -> Result<()>
Configure the cuTENSOR structural permutation plan entry bound.
The cache is initialized if it does not already exist so a setting made before the first CUDA structural permutation call is preserved.
§Errors
Returns crate::Error::RuntimeState if the cache mutex is poisoned.
Trait Implementations§
impl BackendRuntimeCache for CudaBackend
Source§impl BackendSessionHost for CudaBackend
impl BackendSessionHost for CudaBackend
Source§fn with_backend_session<R: Send>(
&mut self,
f: impl FnOnce(&mut dyn BackendSession) -> R + Send,
) -> Result<R, SessionEntryError>
fn with_backend_session<R: Send>( &mut self, f: impl FnOnce(&mut dyn BackendSession) -> R + Send, ) -> Result<R, SessionEntryError>
f inside it. Read moreSource§impl Clone for CudaBackend
impl Clone for CudaBackend
Source§fn clone(&self) -> CudaBackend
fn clone(&self) -> CudaBackend
1.0.0 (const: unstable) · Source§fn clone_from(&mut self, source: &Self)
fn clone_from(&mut self, source: &Self)
source. Read moreSource§impl Debug for CudaBackend
impl Debug for CudaBackend
Source§impl DotGeneralPreparation for CudaBackend
impl DotGeneralPreparation for CudaBackend
Source§fn prepare(
&self,
request: DotGeneralPrepareRequest<'_>,
) -> Result<PrepareCapability, PrepareError>
fn prepare( &self, request: DotGeneralPrepareRequest<'_>, ) -> Result<PrepareCapability, PrepareError>
Source§impl ElementwiseRuntime for CudaBackend
impl ElementwiseRuntime for CudaBackend
Source§fn prepare(
&self,
request: ElementwisePrepareRequest<'_>,
) -> Result<PrepareCapability, PrepareError>
fn prepare( &self, request: ElementwisePrepareRequest<'_>, ) -> Result<PrepareCapability, PrepareError>
Source§impl IndexingRuntime for CudaBackend
impl IndexingRuntime for CudaBackend
Source§fn prepare(
&self,
request: IndexingPrepareRequest<'_>,
) -> Result<PrepareCapability, PrepareError>
fn prepare( &self, request: IndexingPrepareRequest<'_>, ) -> Result<PrepareCapability, PrepareError>
Source§impl LayoutRuntime for CudaBackend
impl LayoutRuntime for CudaBackend
Source§fn prepare(
&self,
request: LayoutPrepareRequest<'_>,
) -> Result<PrepareCapability, PrepareError>
fn prepare( &self, request: LayoutPrepareRequest<'_>, ) -> Result<PrepareCapability, PrepareError>
Source§impl ReductionRuntime for CudaBackend
impl ReductionRuntime for CudaBackend
Source§fn prepare(
&self,
request: ReductionPrepareRequest<'_>,
) -> Result<PrepareCapability, PrepareError>
fn prepare( &self, request: ReductionPrepareRequest<'_>, ) -> Result<PrepareCapability, PrepareError>
Source§impl RuntimeCacheOwner for CudaBackend
impl RuntimeCacheOwner for CudaBackend
Source§fn cache_stats(&self) -> Result<CacheStats, CacheOwnerError>
fn cache_stats(&self) -> Result<CacheStats, CacheOwnerError>
Source§fn clear_caches(&self) -> Result<(), CacheOwnerError>
fn clear_caches(&self) -> Result<(), CacheOwnerError>
impl TensorBackend for CudaBackend
Source§impl TensorBackendCapability for CudaBackend
impl TensorBackendCapability for CudaBackend
fn backend_id(&self) -> BackendId
fn capabilities(&self) -> &'static [OperationCapability]
Source§fn capability(&self, query: CapabilityQuery) -> Option<OperationCapability>
fn capability(&self, query: CapabilityQuery) -> Option<OperationCapability>
Source§fn require_capability(
&self,
query: CapabilityQuery,
axis: CapabilityAxis,
) -> Result<OperationCapability, Error>
fn require_capability( &self, query: CapabilityQuery, axis: CapabilityAxis, ) -> Result<OperationCapability, Error>
Source§impl TensorDeviceTransfer for CudaBackend
impl TensorDeviceTransfer for CudaBackend
Source§fn download_to_host(&mut self, tensor: TensorRead<'_>) -> Result<Tensor>
fn download_to_host(&mut self, tensor: TensorRead<'_>) -> Result<Tensor>
Source§fn upload_host_tensor(&mut self, tensor: TensorRead<'_>) -> Result<Tensor>
fn upload_host_tensor(&mut self, tensor: TensorRead<'_>) -> Result<Tensor>
Auto Trait Implementations§
impl !RefUnwindSafe for CudaBackend
impl !UnwindSafe for CudaBackend
impl Freeze for CudaBackend
impl Send for CudaBackend
impl Sync for CudaBackend
impl Unpin for CudaBackend
impl UnsafeUnpin for CudaBackend
Blanket Implementations§
Source§impl<T> BorrowMut<T> for Twhere
T: ?Sized,
impl<T> BorrowMut<T> for Twhere
T: ?Sized,
Source§fn borrow_mut(&mut self) -> &mut T
fn borrow_mut(&mut self) -> &mut T
impl<ST, DT> CastableFrom<ST, Initialized, Initialized> for DT
impl<ST, DT> CastableFrom<ST, Uninit, Uninit> for DT
§impl<C> CloneExpand for Cwhere
C: Clone,
impl<C> CloneExpand for Cwhere
C: Clone,
fn __expand_clone_method(&self, _scope: &mut Scope) -> C
Source§impl<T> CloneToUninit for Twhere
T: Clone,
impl<T> CloneToUninit for Twhere
T: Clone,
impl<T, U> Imply<T> for U
Source§impl<T> IntoEither for T
impl<T> IntoEither for T
Source§fn into_either(self, into_left: bool) -> Either<Self, Self> ⓘ
fn into_either(self, into_left: bool) -> Either<Self, Self> ⓘ
self into a Left variant of Either<Self, Self>
if into_left is true.
Converts self into a Right variant of Either<Self, Self>
otherwise. Read moreSource§fn into_either_with<F>(self, into_left: F) -> Either<Self, Self> ⓘ
fn into_either_with<F>(self, into_left: F) -> Either<Self, Self> ⓘ
self into a Left variant of Either<Self, Self>
if into_left(&self) returns true.
Converts self into a Right variant of Either<Self, Self>
otherwise. Read more