pub struct CpuBatchThresholds { /* private fields */ }Expand description
Thresholds that CpuBatchStrategy::Auto applies.
The defaults are the values the backend used before these knobs existed; they are not tuned results.
§Examples
use tenferro_cpu::CpuBatchThresholds;
let thresholds = CpuBatchThresholds::default()
.with_vendor_batch_max_item_dim(8)
.with_outer_min_items(4);
assert_eq!(thresholds.vendor_batch_max_item_dim(), 8);
assert_eq!(thresholds.outer_min_items(), 4);
assert_eq!(thresholds.outer_min_items_per_lane(), 1);Implementations§
Source§impl CpuBatchThresholds
impl CpuBatchThresholds
Sourcepub fn vendor_batch_max_item_dim(&self) -> usize
pub fn vendor_batch_max_item_dim(&self) -> usize
Per-item work: the largest m, n and k for which Auto lets a
vendor batch call (cblas_?gemm_batch) handle a grouped GEMM batch.
Strided batched contractions do not use this cutoff: under Auto they
keep one provider GEMM per item, and reach the vendor batch call only
through CpuBatchStrategy::WholeBatchVendor.
§Examples
assert_eq!(tenferro_cpu::CpuBatchThresholds::default().vendor_batch_max_item_dim(), 16);Sourcepub fn outer_min_items(&self) -> usize
pub fn outer_min_items(&self) -> usize
Total batch work: the fewest items for which Auto fans out.
§Examples
assert_eq!(tenferro_cpu::CpuBatchThresholds::default().outer_min_items(), 2);Sourcepub fn outer_min_items_per_lane(&self) -> usize
pub fn outer_min_items_per_lane(&self) -> usize
Chunk granularity: the fewest items each lane must receive before
Auto fans a batch out over every lane.
§Examples
assert_eq!(tenferro_cpu::CpuBatchThresholds::default().outer_min_items_per_lane(), 1);Sourcepub fn lane_item_overhead_ns(&self) -> usize
pub fn lane_item_overhead_ns(&self) -> usize
Lane cost model: the estimated fixed cost of one GEMM item, in nanoseconds, that a lane must amortize.
§Examples
assert_eq!(tenferro_cpu::CpuBatchThresholds::default().lane_item_overhead_ns(), 50);Sourcepub fn lane_muladds_per_ns(&self) -> usize
pub fn lane_muladds_per_ns(&self) -> usize
Lane cost model: estimated multiply-adds per nanosecond of one lane.
§Examples
assert_eq!(tenferro_cpu::CpuBatchThresholds::default().lane_muladds_per_ns(), 16);Sourcepub fn lane_min_work_ns(&self) -> usize
pub fn lane_min_work_ns(&self) -> usize
Lane cost model: the least estimated work, in nanoseconds, each lane
must receive before Auto splits a batch inside an entered session.
Zero lets the item thresholds alone decide.
§Examples
assert_eq!(tenferro_cpu::CpuBatchThresholds::default().lane_min_work_ns(), 8_000);Sourcepub fn with_lane_item_overhead_ns(self, nanoseconds: usize) -> Self
pub fn with_lane_item_overhead_ns(self, nanoseconds: usize) -> Self
Return these thresholds with a different per-item lane overhead.
§Examples
use tenferro_cpu::CpuBatchThresholds;
let thresholds = CpuBatchThresholds::default().with_lane_item_overhead_ns(100);
assert_eq!(thresholds.lane_item_overhead_ns(), 100);Sourcepub fn with_lane_muladds_per_ns(self, muladds: usize) -> Self
pub fn with_lane_muladds_per_ns(self, muladds: usize) -> Self
Return these thresholds with a different lane throughput estimate. Zero is treated as one multiply-add per nanosecond.
§Examples
use tenferro_cpu::CpuBatchThresholds;
let thresholds = CpuBatchThresholds::default().with_lane_muladds_per_ns(32);
assert_eq!(thresholds.lane_muladds_per_ns(), 32);Sourcepub fn with_lane_min_work_ns(self, nanoseconds: usize) -> Self
pub fn with_lane_min_work_ns(self, nanoseconds: usize) -> Self
Return these thresholds with a different minimum work per lane.
§Examples
use tenferro_cpu::CpuBatchThresholds;
let thresholds = CpuBatchThresholds::default().with_lane_min_work_ns(0);
assert_eq!(thresholds.lane_min_work_ns(), 0);Sourcepub fn with_vendor_batch_max_item_dim(self, limit: usize) -> Self
pub fn with_vendor_batch_max_item_dim(self, limit: usize) -> Self
Return these thresholds with a different vendor-batch item limit.
§Examples
use tenferro_cpu::CpuBatchThresholds;
let thresholds = CpuBatchThresholds::default().with_vendor_batch_max_item_dim(0);
assert_eq!(thresholds.vendor_batch_max_item_dim(), 0);Sourcepub fn with_outer_min_items(self, items: usize) -> Self
pub fn with_outer_min_items(self, items: usize) -> Self
Return these thresholds with a different minimum fan-out batch size.
§Examples
use tenferro_cpu::CpuBatchThresholds;
let thresholds = CpuBatchThresholds::default().with_outer_min_items(64);
assert_eq!(thresholds.outer_min_items(), 64);Sourcepub fn with_outer_min_items_per_lane(self, items: usize) -> Self
pub fn with_outer_min_items_per_lane(self, items: usize) -> Self
Return these thresholds with a different minimum chunk per lane.
§Examples
use tenferro_cpu::CpuBatchThresholds;
let thresholds = CpuBatchThresholds::default().with_outer_min_items_per_lane(8);
assert_eq!(thresholds.outer_min_items_per_lane(), 8);Sourcepub fn fans_out(&self, items: usize, lanes: usize) -> bool
pub fn fans_out(&self, items: usize, lanes: usize) -> bool
Whether Auto fans items out over lanes lanes: more than one lane,
at least Self::outer_min_items items, and at least
Self::outer_min_items_per_lane items for every lane.
§Examples
use tenferro_cpu::CpuBatchThresholds;
let thresholds = CpuBatchThresholds::default();
assert!(thresholds.fans_out(8, 4));
assert!(!thresholds.fans_out(3, 4));
assert!(!thresholds.fans_out(8, 1));Trait Implementations§
Source§impl Clone for CpuBatchThresholds
impl Clone for CpuBatchThresholds
Source§fn clone(&self) -> CpuBatchThresholds
fn clone(&self) -> CpuBatchThresholds
1.0.0 (const: unstable) · Source§fn clone_from(&mut self, source: &Self)
fn clone_from(&mut self, source: &Self)
source. Read moreimpl Copy for CpuBatchThresholds
Source§impl Debug for CpuBatchThresholds
impl Debug for CpuBatchThresholds
Source§impl Default for CpuBatchThresholds
impl Default for CpuBatchThresholds
impl Eq for CpuBatchThresholds
Source§impl Hash for CpuBatchThresholds
impl Hash for CpuBatchThresholds
Source§impl PartialEq for CpuBatchThresholds
impl PartialEq for CpuBatchThresholds
impl StructuralPartialEq for CpuBatchThresholds
Auto Trait Implementations§
impl Freeze for CpuBatchThresholds
impl RefUnwindSafe for CpuBatchThresholds
impl Send for CpuBatchThresholds
impl Sync for CpuBatchThresholds
impl Unpin for CpuBatchThresholds
impl UnsafeUnpin for CpuBatchThresholds
impl UnwindSafe for CpuBatchThresholds
Blanket Implementations§
impl<T> Boilerplate for T
Source§impl<T> BorrowMut<T> for Twhere
T: ?Sized,
impl<T> BorrowMut<T> for Twhere
T: ?Sized,
Source§fn borrow_mut(&mut self) -> &mut T
fn borrow_mut(&mut self) -> &mut T
Source§impl<T> CloneToUninit for Twhere
T: Clone,
impl<T> CloneToUninit for Twhere
T: Clone,
§impl<Q, K> Equivalent<K> for Q
impl<Q, K> Equivalent<K> for Q
§fn equivalent(&self, key: &K) -> bool
fn equivalent(&self, key: &K) -> bool
impl<T, U> Imply<T> for U
Source§impl<T> IntoEither for T
impl<T> IntoEither for T
Source§fn into_either(self, into_left: bool) -> Either<Self, Self> ⓘ
fn into_either(self, into_left: bool) -> Either<Self, Self> ⓘ
self into a Left variant of Either<Self, Self>
if into_left is true.
Converts self into a Right variant of Either<Self, Self>
otherwise. Read moreSource§fn into_either_with<F>(self, into_left: F) -> Either<Self, Self> ⓘ
fn into_either_with<F>(self, into_left: F) -> Either<Self, Self> ⓘ
self into a Left variant of Either<Self, Self>
if into_left(&self) returns true.
Converts self into a Right variant of Either<Self, Self>
otherwise. Read more