Skip to main content

tenferro_cpu/
placement.rs

1use thiserror::Error;
2
3use crate::{CpuBackendKind, CpuContextError, CpuSet, CpuTopology, CpuTopologyError, NumaNodeId};
4
5/// Typed failure raised while constructing a CPU execution engine.
6///
7/// The tensor-backed compatibility path and the managed engine path expose
8/// different concrete construction errors. This wrapper keeps both sources
9/// typed while allowing [`CpuPlacementError`] to present one public error
10/// shape.
11///
12/// # Examples
13///
14/// ```
15/// use tenferro_cpu::{CpuEngineConstructionError, CpuContextError};
16/// use std::error::Error;
17///
18/// let error = CpuEngineConstructionError::Context(CpuContextError::InvalidThreadCount);
19/// assert!(error.source().is_some());
20/// ```
21#[derive(Debug, Error)]
22pub enum CpuEngineConstructionError {
23    /// A managed CPU context or pinned worker engine could not be built.
24    #[error("managed CPU engine construction failed: {0}")]
25    Context(#[source] CpuContextError),
26    /// The tensor-backed compatibility engine could not be built.
27    #[error("tensor CPU engine construction failed: {0}")]
28    Tensor(#[source] tenferro_tensor::Error),
29}
30
31/// Requested CPU execution placement.
32///
33/// `AllAllowed` means all logical CPUs permitted by the process affinity mask,
34/// not every CPU installed in the host.
35///
36/// # Examples
37///
38/// ```
39/// use tenferro_cpu::{CpuPlacement, NumaNodeId};
40///
41/// let placement = CpuPlacement::NumaNode(NumaNodeId::new(2));
42/// assert!(matches!(placement, CpuPlacement::NumaNode(_)));
43/// ```
44#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Hash)]
45pub enum CpuPlacement {
46    /// Let the selected provider choose its compatible default policy.
47    #[default]
48    Auto,
49    /// Restrict managed tenferro/faer execution to one usable OS NUMA node.
50    NumaNode(NumaNodeId),
51    /// Use the complete CPU set permitted to the process.
52    AllAllowed,
53}
54
55/// Concrete CPU placement resolved for a managed domain or declared by an external domain.
56///
57/// # Examples
58///
59/// ```
60/// use tenferro_cpu::{CpuId, CpuSet, ResolvedCpuPlacement};
61///
62/// let placement = ResolvedCpuPlacement::AllAllowed {
63///     cpus: CpuSet::new([CpuId::new(0)])?,
64/// };
65/// assert_eq!(placement.cpus().len(), 1);
66/// # Ok::<(), tenferro_cpu::CpuSetError>(())
67/// ```
68#[derive(Clone, Debug, PartialEq, Eq)]
69pub enum ResolvedCpuPlacement {
70    /// A concrete OS NUMA-node placement.
71    NumaNode {
72        /// The sparse OS NUMA node ID.
73        id: NumaNodeId,
74        /// The logical CPUs resolved or declared for the node.
75        cpus: CpuSet,
76    },
77    /// A resolved or declared complete process-affinity CPU set.
78    AllAllowed {
79        /// Logical CPUs resolved or declared as process-permitted.
80        cpus: CpuSet,
81    },
82}
83
84impl ResolvedCpuPlacement {
85    /// Return the concrete logical CPU set resolved or declared for this placement.
86    ///
87    /// # Examples
88    ///
89    /// ```
90    /// use tenferro_cpu::{CpuId, CpuSet, ResolvedCpuPlacement};
91    ///
92    /// let placement = ResolvedCpuPlacement::AllAllowed {
93    ///     cpus: CpuSet::new([CpuId::new(1), CpuId::new(2)])?,
94    /// };
95    /// assert_eq!(placement.cpus().as_usize_vec(), vec![1, 2]);
96    /// # Ok::<(), tenferro_cpu::CpuSetError>(())
97    /// ```
98    pub fn cpus(&self) -> &CpuSet {
99        match self {
100            Self::NumaNode { cpus, .. } | Self::AllAllowed { cpus } => cpus,
101        }
102    }
103
104    /// Return the OS NUMA node ID for a node placement.
105    ///
106    /// # Examples
107    ///
108    /// ```
109    /// use tenferro_cpu::{CpuId, CpuSet, NumaNodeId, ResolvedCpuPlacement};
110    ///
111    /// let placement = ResolvedCpuPlacement::NumaNode {
112    ///     id: NumaNodeId::new(7),
113    ///     cpus: CpuSet::new([CpuId::new(3)])?,
114    /// };
115    /// assert_eq!(placement.node_id(), Some(NumaNodeId::new(7)));
116    /// # Ok::<(), tenferro_cpu::CpuSetError>(())
117    /// ```
118    pub fn node_id(&self) -> Option<NumaNodeId> {
119        match self {
120            Self::NumaNode { id, .. } => Some(*id),
121            Self::AllAllowed { .. } => None,
122        }
123    }
124}
125
126/// Failure to resolve a CPU placement for the selected public provider kind.
127///
128/// # Examples
129///
130/// ```
131/// use tenferro_cpu::{CpuBackendKind, CpuPlacement, CpuPlacementError};
132///
133/// let error = CpuPlacementError::ExternalProviderAffinityUnmanaged {
134///     requested: CpuPlacement::AllAllowed,
135///     backend: CpuBackendKind::Blas,
136/// };
137/// assert!(error.to_string().contains("affinity"));
138/// ```
139#[derive(Debug, Error)]
140pub enum CpuPlacementError {
141    /// Process-visible topology discovery failed before placement resolution.
142    #[error("cannot resolve {requested:?} for {backend:?}: topology discovery failed: {source}")]
143    TopologyDiscovery {
144        /// The placement requested by the caller.
145        requested: CpuPlacement,
146        /// The selected public backend kind.
147        backend: CpuBackendKind,
148        /// The preserved topology failure category.
149        #[source]
150        source: CpuTopologyError,
151    },
152    /// The current platform cannot construct verified pinned worker pools.
153    #[error(
154        "cannot resolve {requested:?} for {backend:?}: managed worker affinity is unavailable"
155    )]
156    ManagedAffinityUnavailable {
157        /// The explicit placement requested by the caller.
158        requested: CpuPlacement,
159        /// The selected public backend kind.
160        backend: CpuBackendKind,
161    },
162    /// NUMA-node placement was requested but OS NUMA discovery was unavailable.
163    #[error("cannot resolve {requested:?} for {backend:?}: NUMA discovery is unavailable")]
164    NumaDiscoveryUnavailable {
165        /// The placement requested by the caller.
166        requested: CpuPlacement,
167        /// The selected public backend kind.
168        backend: CpuBackendKind,
169    },
170    /// The requested OS NUMA node has no usable CPUs in this process.
171    #[error("cannot resolve {requested:?} for {backend:?}: NUMA node {node} is unavailable")]
172    UnknownNumaNode {
173        /// The placement requested by the caller.
174        requested: CpuPlacement,
175        /// The selected public backend kind.
176        backend: CpuBackendKind,
177        /// The unknown or process-unavailable OS node ID.
178        node: NumaNodeId,
179    },
180    /// An external provider owns worker affinity, so explicit placement is unsafe.
181    #[error(
182        "cannot resolve {requested:?} for {backend:?}: external provider worker affinity is unmanaged"
183    )]
184    ExternalProviderAffinityUnmanaged {
185        /// The explicit placement requested by the caller.
186        requested: CpuPlacement,
187        /// The selected public backend kind.
188        backend: CpuBackendKind,
189    },
190    /// An externally managed coordinator has no domain for the explicit placement.
191    #[error("externally managed CPU coordinator has no registered domain for {requested:?}")]
192    UnregisteredExternalPlacement {
193        /// The explicit registry-only placement request.
194        requested: CpuPlacement,
195    },
196    /// An externally managed coordinator has no domain with the requested ID.
197    #[error("externally managed CPU coordinator has no registered domain {domain:?}")]
198    UnregisteredExternalDomain {
199        /// Missing caller-stable domain identity.
200        domain: crate::CpuDomainId,
201    },
202    /// A pinned engine could not be built for an otherwise valid placement.
203    #[error("cannot resolve {requested:?} for {backend:?}: engine construction failed: {source}")]
204    EngineConstruction {
205        /// The placement requested by the caller.
206        requested: CpuPlacement,
207        /// The selected public backend kind.
208        backend: CpuBackendKind,
209        /// Typed worker-pool construction or affinity failure.
210        #[source]
211        source: CpuEngineConstructionError,
212    },
213    /// The placement state reached an impossible internal compatibility mode.
214    #[error("cannot resolve {requested:?} for {backend:?}: {message}")]
215    InternalState {
216        /// The placement requested by the caller.
217        requested: CpuPlacement,
218        /// The selected public backend kind.
219        backend: CpuBackendKind,
220        /// Stable internal-state diagnostic.
221        message: &'static str,
222    },
223}
224
225#[derive(Clone, Debug, PartialEq, Eq)]
226pub(crate) enum ResolvedCpuExecution {
227    Compatibility,
228    Managed(ResolvedCpuPlacement),
229    ExternalManaged(ResolvedCpuPlacement),
230    ExternalCallerManaged,
231    ProviderDefaultExclusive,
232}
233
234pub(crate) fn resolve_placement(
235    backend: CpuBackendKind,
236    requested: CpuPlacement,
237    topology: &CpuTopology,
238) -> Result<ResolvedCpuExecution, CpuPlacementError> {
239    resolve_placement_with_affinity(
240        backend,
241        requested,
242        topology,
243        cfg!(any(target_os = "linux", target_os = "android")),
244    )
245}
246
247pub(crate) fn resolve_placement_with_affinity(
248    backend: CpuBackendKind,
249    requested: CpuPlacement,
250    topology: &CpuTopology,
251    managed_affinity_available: bool,
252) -> Result<ResolvedCpuExecution, CpuPlacementError> {
253    if backend == CpuBackendKind::Blas {
254        return match requested {
255            CpuPlacement::Auto => Ok(ResolvedCpuExecution::ProviderDefaultExclusive),
256            CpuPlacement::NumaNode(_) | CpuPlacement::AllAllowed => {
257                Err(CpuPlacementError::ExternalProviderAffinityUnmanaged { requested, backend })
258            }
259        };
260    }
261
262    if !managed_affinity_available {
263        return match requested {
264            CpuPlacement::Auto => Ok(ResolvedCpuExecution::Compatibility),
265            CpuPlacement::NumaNode(_) | CpuPlacement::AllAllowed => {
266                Err(CpuPlacementError::ManagedAffinityUnavailable { requested, backend })
267            }
268        };
269    }
270
271    let placement = match requested {
272        CpuPlacement::Auto | CpuPlacement::AllAllowed => ResolvedCpuPlacement::AllAllowed {
273            cpus: topology.allowed_cpus().clone(),
274        },
275        CpuPlacement::NumaNode(node) => {
276            if !topology.has_numa_nodes() {
277                return Err(CpuPlacementError::NumaDiscoveryUnavailable { requested, backend });
278            }
279            let cpus = topology
280                .node(node)
281                .ok_or(CpuPlacementError::UnknownNumaNode {
282                    requested,
283                    backend,
284                    node,
285                })?;
286            ResolvedCpuPlacement::NumaNode {
287                id: node,
288                cpus: cpus.cpus().clone(),
289            }
290        }
291    };
292    Ok(ResolvedCpuExecution::Managed(placement))
293}
294
295#[cfg(test)]
296mod tests;