tenferro_linalg/lib.rs
1//! Linear algebra extension operations for tenferro.
2//!
3//! This crate owns the graph-facing linalg op payloads and runtime
4//! registration. Tensor-facing operations are exposed through extension traits.
5//! CPU backend kernels live in this crate behind the linalg backend trait.
6//! A CPU backend paired by `tenferro_gpu::apple::AppleContext` additionally supports
7//! guarded rank-2 Cholesky on matching Apple managed `F32`, `F64`, `C32`, and
8//! `C64` tensors. This is an explicit CPU selection and is not a general
9//! managed-memory fallback for other linalg operations.
10//!
11//! # Examples
12//!
13//! ```
14//! use tenferro_linalg::TracedTensorLinalgExt;
15//! use tenferro_cpu::CpuBackend;
16//! use tenferro_runtime::{GraphCompiler, Runtime, TracedTensor};
17//!
18//! let a = TracedTensor::from_vec_col_major(
19//! vec![2, 2],
20//! vec![4.0_f64, 2.0, 2.0, 3.0],
21//! )
22//! .unwrap();
23//! let l = a.cholesky().unwrap();
24//!
25//! let mut compiler = GraphCompiler::new();
26//! let program = compiler.compile(&l).unwrap();
27//! let backend = CpuBackend::new();
28//! let engine_id = tenferro_cpu::runtime_engine_id().unwrap();
29//! let mut builder = Runtime::builder();
30//! builder
31//! .register_engine(tenferro_cpu::runtime_engine_registration(&backend).unwrap())
32//! .unwrap();
33//! builder
34//! .install_extension_module(tenferro_linalg::extension_module::<CpuBackend>(engine_id).unwrap())
35//! .unwrap();
36//! let runtime = builder.build().unwrap();
37//! let out = runtime.run_compiled(&program, &[]).unwrap().pop().unwrap();
38//! assert_eq!(out.shape(), &[2, 2]);
39//! ```
40//!
41//! # Cargo features
42//!
43//! | Feature | Enables |
44//! |---|---|
45//! | `cpu-faer` (default) and the other CPU provider features | Forwarded to `tenferro-cpu`; see its documentation. |
46//! | `autodiff` | The eager surface (`EagerSessionLinalgExt`, `EagerTensorLinalgExt`) and AD rules. Adds the `tenferro-ad` dependency. |
47//! | `cuda` | CUDA execution through `tenferro-gpu`. |
48//! | `webgpu` | WebGPU/Metal execution through `tenferro-gpu` (a subset of operations). |
49//! | `rocm` | Placeholder; HIP/ROCm is not implemented. |
50//!
51//! `autodiff` is not a heavy dependency switch: `tenferro-ad` is the crate that
52//! owns `EagerSession`/`EagerTensor`, so an eager linalg surface without it
53//! would save nothing and would only produce eager tensors whose linalg ops
54//! fail on backward (#1972). For inference without AD, call the same
55//! operations on concrete tensors inside a backend session through
56//! [`TensorLinalgExt`] / [`TypedTensorLinalgExt`], which need no `autodiff`:
57//!
58//! ```
59//! use tenferro_cpu::CpuBackend;
60//! use tenferro_linalg::TypedTensorLinalgExt;
61//! use tenferro_tensor::{BackendSessionHost, TypedTensor};
62//!
63//! let mut backend = CpuBackend::new();
64//! // Lower-triangular [[2, 0], [1, 1]] (column-major) and right-hand side.
65//! let l = TypedTensor::<f64>::from_vec_col_major(vec![2, 2], vec![2.0, 1.0, 0.0, 1.0])?;
66//! let b = TypedTensor::<f64>::from_vec_col_major(vec![2, 1], vec![4.0, 5.0])?;
67//! let x = backend.with_backend_session(|session| {
68//! // left_side, lower, transpose_a, unit_diagonal
69//! l.triangular_solve(&b, true, true, false, false, session)
70//! })??;
71//! assert_eq!(x.host_data()?, &[2.0, 3.0]);
72//! # Ok::<(), Box<dyn std::error::Error>>(())
73//! ```
74// A misaligned pointer handed to a CUDA library is undefined behaviour and
75// fails only on some library versions: `cuDoubleComplex` is `double2`, which
76// CUDA declares `__align__(16)`, while `num_complex::Complex64` is 8-aligned,
77// so casting `&Complex64` to `*const cuDoubleComplex` produced a pointer
78// cuBLAS >= 12.9 faults on (issue #1870). Deny the whole cast class at this
79// FFI boundary rather than re-auditing it by hand.
80#![cfg_attr(docsrs, feature(doc_cfg))]
81#![deny(clippy::cast_ptr_alignment)]
82
83#[cfg(feature = "autodiff")]
84mod ad;
85pub mod backend;
86mod cpu;
87pub mod cpu_kernels;
88#[cfg(feature = "autodiff")]
89mod eager_composites;
90#[cfg(feature = "autodiff")]
91mod eager_ext;
92pub mod error;
93mod extension;
94#[cfg(feature = "cuda")]
95mod gpu;
96mod householder;
97pub mod prelude;
98mod rank_revealing_qr;
99mod tensor_ext;
100mod traced;
101mod validation;
102
103#[cfg(feature = "autodiff")]
104pub use ad::semantic_ad_rules;
105#[cfg(feature = "autodiff")]
106pub use ad::support::{
107 all_linalg_ad_support, linalg_ad_support, LinalgAdModeSupport, LinalgAdOpKind,
108 LinalgAdOutputSupport, LinalgAdRoute, LinalgAdRuleSupport, LinalgAdSupport,
109};
110pub use backend::LinalgBackend;
111#[cfg(feature = "autodiff")]
112#[cfg_attr(docsrs, doc(cfg(feature = "autodiff")))]
113pub use eager_ext::{EagerSessionLinalgExt, EagerTensorLinalgExt};
114pub use error::{Error, Result};
115pub use extension::{
116 extension_module, EighDriver, EighGauge, EighOptions, QrGauge, QrOptions, SvdDriver, SvdGauge,
117 SvdOptions, DEFAULT_DECOMPOSITION_DERIVATIVE_EPS, LINALG_EXTENSION_FAMILY_ID,
118};
119pub use householder::HouseholderQr;
120pub use rank_revealing_qr::{RankRevealingQrOptions, RankRevealingQrResult};
121pub use tensor_ext::{
122 LinalgScalar, TensorLinalgExt, TensorReadLinalgExt, TypedEig, TypedFullPivLu, TypedLu,
123 TypedRankRevealingQrResult, TypedSvd, TypedTensorLinalgExt,
124};
125pub use traced::TracedTensorLinalgExt;